billion-context 0.1.87 → 0.1.88-pr.571.415

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -9365,7 +9365,7 @@ var require_pool = __commonJS({
9365
9365
  function defaultFactory(origin, opts) {
9366
9366
  return new Client(origin, opts);
9367
9367
  }
9368
- var Pool = class extends PoolBase {
9368
+ var Pool2 = class extends PoolBase {
9369
9369
  constructor(origin, {
9370
9370
  connections,
9371
9371
  factory = defaultFactory,
@@ -9438,7 +9438,7 @@ var require_pool = __commonJS({
9438
9438
  }
9439
9439
  }
9440
9440
  };
9441
- module.exports = Pool;
9441
+ module.exports = Pool2;
9442
9442
  }
9443
9443
  });
9444
9444
 
@@ -9458,7 +9458,7 @@ var require_balanced_pool = __commonJS({
9458
9458
  kRemoveClient,
9459
9459
  kGetDispatcher
9460
9460
  } = require_pool_base();
9461
- var Pool = require_pool();
9461
+ var Pool2 = require_pool();
9462
9462
  var { kUrl } = require_symbols();
9463
9463
  var util = require_util();
9464
9464
  var kFactory = /* @__PURE__ */ Symbol("factory");
@@ -9479,7 +9479,7 @@ var require_balanced_pool = __commonJS({
9479
9479
  return a;
9480
9480
  }
9481
9481
  function defaultFactory(origin, opts) {
9482
- return new Pool(origin, opts);
9482
+ return new Pool2(origin, opts);
9483
9483
  }
9484
9484
  var BalancedPool = class extends PoolBase {
9485
9485
  constructor(upstreams = [], { factory = defaultFactory, ...opts } = {}) {
@@ -9714,7 +9714,7 @@ var require_agent = __commonJS({
9714
9714
  var { InvalidArgumentError, MaxOriginsReachedError } = require_errors();
9715
9715
  var { kClients, kRunning, kClose, kDestroy, kDispatch, kUrl } = require_symbols();
9716
9716
  var DispatcherBase = require_dispatcher_base();
9717
- var Pool = require_pool();
9717
+ var Pool2 = require_pool();
9718
9718
  var Client = require_client();
9719
9719
  var util = require_util();
9720
9720
  var kOnConnect = /* @__PURE__ */ Symbol("onConnect");
@@ -9725,9 +9725,9 @@ var require_agent = __commonJS({
9725
9725
  var kOptions = /* @__PURE__ */ Symbol("options");
9726
9726
  var kOrigins = /* @__PURE__ */ Symbol("origins");
9727
9727
  function defaultFactory(origin, opts) {
9728
- return opts && opts.connections === 1 ? new Client(origin, opts) : new Pool(origin, opts);
9728
+ return opts && opts.connections === 1 ? new Client(origin, opts) : new Pool2(origin, opts);
9729
9729
  }
9730
- var Agent = class extends DispatcherBase {
9730
+ var Agent2 = class extends DispatcherBase {
9731
9731
  constructor({ factory = defaultFactory, maxOrigins = Infinity, connect, ...options } = {}) {
9732
9732
  if (typeof factory !== "function") {
9733
9733
  throw new InvalidArgumentError("factory must be a function.");
@@ -9836,7 +9836,7 @@ var require_agent = __commonJS({
9836
9836
  return allClientStats;
9837
9837
  }
9838
9838
  };
9839
- module.exports = Agent;
9839
+ module.exports = Agent2;
9840
9840
  }
9841
9841
  });
9842
9842
 
@@ -10344,7 +10344,7 @@ var require_socks5_proxy_agent = __commonJS({
10344
10344
  var { InvalidArgumentError } = require_errors();
10345
10345
  var { Socks5Client, STATES } = require_socks5_client();
10346
10346
  var { kDispatch, kClose, kDestroy } = require_symbols();
10347
- var Pool = require_pool();
10347
+ var Pool2 = require_pool();
10348
10348
  var buildConnector = require_connect();
10349
10349
  var { debuglog } = __require("util");
10350
10350
  var debug = debuglog("undici:socks5-proxy");
@@ -10467,7 +10467,7 @@ var require_socks5_proxy_agent = __commonJS({
10467
10467
  const originKey = String(origin);
10468
10468
  let pool = this[kPools].get(originKey);
10469
10469
  if (!pool || pool.destroyed || pool.closed) {
10470
- pool = new Pool(origin, {
10470
+ pool = new Pool2(origin, {
10471
10471
  pipelining: opts.pipelining,
10472
10472
  connections: opts.connections,
10473
10473
  connect: async (connectOpts, callback) => {
@@ -10542,8 +10542,8 @@ var require_proxy_agent = __commonJS({
10542
10542
  "node_modules/undici/lib/dispatcher/proxy-agent.js"(exports, module) {
10543
10543
  "use strict";
10544
10544
  var { kProxy, kClose, kDestroy, kDispatch } = require_symbols();
10545
- var Agent = require_agent();
10546
- var Pool = require_pool();
10545
+ var Agent2 = require_agent();
10546
+ var Pool2 = require_pool();
10547
10547
  var DispatcherBase = require_dispatcher_base();
10548
10548
  var { InvalidArgumentError, RequestAbortedError, SecureProxyConnectionError } = require_errors();
10549
10549
  var buildConnector = require_connect();
@@ -10561,7 +10561,7 @@ var require_proxy_agent = __commonJS({
10561
10561
  return protocol === "https:" ? 443 : 80;
10562
10562
  }
10563
10563
  function defaultFactory(origin, opts) {
10564
- return new Pool(origin, opts);
10564
+ return new Pool2(origin, opts);
10565
10565
  }
10566
10566
  var noop = () => {
10567
10567
  };
@@ -10569,7 +10569,7 @@ var require_proxy_agent = __commonJS({
10569
10569
  if (opts.connections === 1) {
10570
10570
  return new Client(origin, opts);
10571
10571
  }
10572
- return new Pool(origin, opts);
10572
+ return new Pool2(origin, opts);
10573
10573
  }
10574
10574
  var Http1ProxyWrapper = class extends DispatcherBase {
10575
10575
  #client;
@@ -10673,7 +10673,7 @@ var require_proxy_agent = __commonJS({
10673
10673
  } else {
10674
10674
  this[kClient] = clientFactory(url, { connect });
10675
10675
  }
10676
- this[kAgent] = new Agent({
10676
+ this[kAgent] = new Agent2({
10677
10677
  ...opts,
10678
10678
  factory,
10679
10679
  connect: async (opts2, callback) => {
@@ -10801,7 +10801,7 @@ var require_env_http_proxy_agent = __commonJS({
10801
10801
  var DispatcherBase = require_dispatcher_base();
10802
10802
  var { kClose, kDestroy, kClosed, kDestroyed, kDispatch, kNoProxyAgent, kHttpProxyAgent, kHttpsProxyAgent } = require_symbols();
10803
10803
  var ProxyAgent2 = require_proxy_agent();
10804
- var Agent = require_agent();
10804
+ var Agent2 = require_agent();
10805
10805
  var DEFAULT_PORTS = {
10806
10806
  "http:": 80,
10807
10807
  "https:": 443
@@ -10814,7 +10814,7 @@ var require_env_http_proxy_agent = __commonJS({
10814
10814
  super();
10815
10815
  this.#opts = opts;
10816
10816
  const { httpProxy, httpsProxy, noProxy, ...agentOpts } = opts;
10817
- this[kNoProxyAgent] = new Agent(agentOpts);
10817
+ this[kNoProxyAgent] = new Agent2(agentOpts);
10818
10818
  const HTTP_PROXY = httpProxy ?? process.env.http_proxy ?? process.env.HTTP_PROXY;
10819
10819
  if (HTTP_PROXY) {
10820
10820
  this[kHttpProxyAgent] = new ProxyAgent2({ ...agentOpts, uri: HTTP_PROXY });
@@ -13384,7 +13384,7 @@ var require_mock_pool = __commonJS({
13384
13384
  "node_modules/undici/lib/mock/mock-pool.js"(exports, module) {
13385
13385
  "use strict";
13386
13386
  var { promisify: promisify2 } = __require("util");
13387
- var Pool = require_pool();
13387
+ var Pool2 = require_pool();
13388
13388
  var { buildMockDispatch } = require_mock_utils();
13389
13389
  var {
13390
13390
  kDispatches,
@@ -13399,7 +13399,7 @@ var require_mock_pool = __commonJS({
13399
13399
  var { MockInterceptor } = require_mock_interceptor();
13400
13400
  var Symbols = require_symbols();
13401
13401
  var { InvalidArgumentError } = require_errors();
13402
- var MockPool = class extends Pool {
13402
+ var MockPool = class extends Pool2 {
13403
13403
  constructor(origin, opts) {
13404
13404
  if (!opts || !opts.agent || typeof opts.agent.dispatch !== "function") {
13405
13405
  throw new InvalidArgumentError("Argument opts.agent must implement Agent");
@@ -13486,7 +13486,7 @@ var require_mock_agent = __commonJS({
13486
13486
  "node_modules/undici/lib/mock/mock-agent.js"(exports, module) {
13487
13487
  "use strict";
13488
13488
  var { kClients } = require_symbols();
13489
- var Agent = require_agent();
13489
+ var Agent2 = require_agent();
13490
13490
  var {
13491
13491
  kAgent,
13492
13492
  kMockAgentSet,
@@ -13524,7 +13524,7 @@ var require_mock_agent = __commonJS({
13524
13524
  if (opts?.agent && typeof opts.agent.dispatch !== "function") {
13525
13525
  throw new InvalidArgumentError("Argument opts.agent must implement Agent");
13526
13526
  }
13527
- const agent = opts?.agent ? opts.agent : new Agent(opts);
13527
+ const agent = opts?.agent ? opts.agent : new Agent2(opts);
13528
13528
  this[kAgent] = agent;
13529
13529
  this[kClients] = agent[kClients];
13530
13530
  this[kOptions] = mockOptions;
@@ -14131,7 +14131,7 @@ var require_snapshot_recorder = __commonJS({
14131
14131
  var require_snapshot_agent = __commonJS({
14132
14132
  "node_modules/undici/lib/mock/snapshot-agent.js"(exports, module) {
14133
14133
  "use strict";
14134
- var Agent = require_agent();
14134
+ var Agent2 = require_agent();
14135
14135
  var MockAgent = require_mock_agent();
14136
14136
  var { SnapshotRecorder } = require_snapshot_recorder();
14137
14137
  var WrapHandler = require_wrap_handler();
@@ -14182,7 +14182,7 @@ var require_snapshot_agent = __commonJS({
14182
14182
  });
14183
14183
  this[kSnapshotLoaded] = false;
14184
14184
  if (this[kSnapshotMode] === "record" || this[kSnapshotMode] === "update" || this[kSnapshotMode] === "playback" && opts.excludeUrls && opts.excludeUrls.length > 0) {
14185
- this[kRealAgent] = new Agent(opts);
14185
+ this[kRealAgent] = new Agent2(opts);
14186
14186
  }
14187
14187
  if ((this[kSnapshotMode] === "playback" || this[kSnapshotMode] === "update") && this[kSnapshotPath]) {
14188
14188
  this.loadSnapshots().catch(() => {
@@ -14421,9 +14421,9 @@ var require_global2 = __commonJS({
14421
14421
  var globalDispatcher = /* @__PURE__ */ Symbol.for("undici.globalDispatcher.2");
14422
14422
  var legacyGlobalDispatcher = /* @__PURE__ */ Symbol.for("undici.globalDispatcher.1");
14423
14423
  var { InvalidArgumentError } = require_errors();
14424
- var Agent = require_agent();
14424
+ var Agent2 = require_agent();
14425
14425
  if (getGlobalDispatcher() === void 0) {
14426
- setGlobalDispatcher(new Agent());
14426
+ setGlobalDispatcher(new Agent2());
14427
14427
  }
14428
14428
  function setGlobalDispatcher(agent) {
14429
14429
  if (!agent || typeof agent.dispatch !== "function") {
@@ -18828,7 +18828,7 @@ var require_headers = __commonJS({
18828
18828
  }
18829
18829
  }
18830
18830
  };
18831
- var Headers = class _Headers {
18831
+ var Headers2 = class _Headers {
18832
18832
  #guard;
18833
18833
  /**
18834
18834
  * @type {HeadersList}
@@ -18969,13 +18969,13 @@ var require_headers = __commonJS({
18969
18969
  target.#headersList = list;
18970
18970
  }
18971
18971
  };
18972
- var { getHeadersGuard, setHeadersGuard, getHeadersList, setHeadersList } = Headers;
18973
- Reflect.deleteProperty(Headers, "getHeadersGuard");
18974
- Reflect.deleteProperty(Headers, "setHeadersGuard");
18975
- Reflect.deleteProperty(Headers, "getHeadersList");
18976
- Reflect.deleteProperty(Headers, "setHeadersList");
18977
- iteratorMixin("Headers", Headers, headersListSortAndCombine, 0, 1);
18978
- Object.defineProperties(Headers.prototype, {
18972
+ var { getHeadersGuard, setHeadersGuard, getHeadersList, setHeadersList } = Headers2;
18973
+ Reflect.deleteProperty(Headers2, "getHeadersGuard");
18974
+ Reflect.deleteProperty(Headers2, "setHeadersGuard");
18975
+ Reflect.deleteProperty(Headers2, "getHeadersList");
18976
+ Reflect.deleteProperty(Headers2, "setHeadersList");
18977
+ iteratorMixin("Headers", Headers2, headersListSortAndCombine, 0, 1);
18978
+ Object.defineProperties(Headers2.prototype, {
18979
18979
  append: kEnumerableProperty,
18980
18980
  delete: kEnumerableProperty,
18981
18981
  get: kEnumerableProperty,
@@ -18993,7 +18993,7 @@ var require_headers = __commonJS({
18993
18993
  webidl.converters.HeadersInit = function(V2, prefix, argument) {
18994
18994
  if (webidl.util.Type(V2) === webidl.util.Types.OBJECT) {
18995
18995
  const iterator = Reflect.get(V2, Symbol.iterator);
18996
- if (!util.types.isProxy(V2) && iterator === Headers.prototype.entries) {
18996
+ if (!util.types.isProxy(V2) && iterator === Headers2.prototype.entries) {
18997
18997
  try {
18998
18998
  return getHeadersList(V2).entriesList;
18999
18999
  } catch {
@@ -19014,7 +19014,7 @@ var require_headers = __commonJS({
19014
19014
  fill: fill2,
19015
19015
  // for test.
19016
19016
  compareHeaderName,
19017
- Headers,
19017
+ Headers: Headers2,
19018
19018
  HeadersList,
19019
19019
  getHeadersGuard,
19020
19020
  setHeadersGuard,
@@ -19028,7 +19028,7 @@ var require_headers = __commonJS({
19028
19028
  var require_response = __commonJS({
19029
19029
  "node_modules/undici/lib/web/fetch/response.js"(exports, module) {
19030
19030
  "use strict";
19031
- var { Headers, HeadersList, fill: fill2, getHeadersGuard, setHeadersGuard, setHeadersList } = require_headers();
19031
+ var { Headers: Headers2, HeadersList, fill: fill2, getHeadersGuard, setHeadersGuard, setHeadersList } = require_headers();
19032
19032
  var { extractBody, cloneBody, mixinBody, streamRegistry, bodyUnusable } = require_body();
19033
19033
  var util = require_util();
19034
19034
  var nodeUtil = __require("util");
@@ -19104,7 +19104,7 @@ var require_response = __commonJS({
19104
19104
  }
19105
19105
  init = webidl.converters.ResponseInit(init);
19106
19106
  this.#state = makeResponse({});
19107
- this.#headers = new Headers(kConstruct);
19107
+ this.#headers = new Headers2(kConstruct);
19108
19108
  setHeadersGuard(this.#headers, "response");
19109
19109
  setHeadersList(this.#headers, this.#state.headersList);
19110
19110
  let bodyWithType = null;
@@ -19379,7 +19379,7 @@ var require_response = __commonJS({
19379
19379
  function fromInnerResponse(innerResponse, guard) {
19380
19380
  const response = new Response2(kConstruct);
19381
19381
  setResponseState(response, innerResponse);
19382
- const headers = new Headers(kConstruct);
19382
+ const headers = new Headers2(kConstruct);
19383
19383
  setResponseHeaders(response, headers);
19384
19384
  setHeadersList(headers, innerResponse.headersList);
19385
19385
  setHeadersGuard(headers, guard);
@@ -19451,7 +19451,7 @@ var require_request2 = __commonJS({
19451
19451
  "node_modules/undici/lib/web/fetch/request.js"(exports, module) {
19452
19452
  "use strict";
19453
19453
  var { extractBody, mixinBody, cloneBody, bodyUnusable } = require_body();
19454
- var { Headers, fill: fillHeaders, HeadersList, setHeadersGuard, getHeadersGuard, setHeadersList, getHeadersList } = require_headers();
19454
+ var { Headers: Headers2, fill: fillHeaders, HeadersList, setHeadersGuard, getHeadersGuard, setHeadersList, getHeadersList } = require_headers();
19455
19455
  var util = require_util();
19456
19456
  var nodeUtil = __require("util");
19457
19457
  var {
@@ -19720,7 +19720,7 @@ var require_request2 = __commonJS({
19720
19720
  requestFinalizer.register(ac, { signal, abort }, abort);
19721
19721
  }
19722
19722
  }
19723
- this.#headers = new Headers(kConstruct);
19723
+ this.#headers = new Headers2(kConstruct);
19724
19724
  setHeadersList(this.#headers, request.headersList);
19725
19725
  setHeadersGuard(this.#headers, "request");
19726
19726
  if (mode === "no-cors") {
@@ -20061,7 +20061,7 @@ var require_request2 = __commonJS({
20061
20061
  setRequestState(request, innerRequest);
20062
20062
  setRequestDispatcher(request, dispatcher);
20063
20063
  setRequestSignal(request, signal);
20064
- const headers = new Headers(kConstruct);
20064
+ const headers = new Headers2(kConstruct);
20065
20065
  setRequestHeaders(request, headers);
20066
20066
  setHeadersList(headers, innerRequest.headersList);
20067
20067
  setHeadersGuard(headers, guard);
@@ -22538,8 +22538,8 @@ var require_cookies = __commonJS({
22538
22538
  var { parseSetCookie } = require_parse();
22539
22539
  var { stringify } = require_util4();
22540
22540
  var { webidl } = require_webidl();
22541
- var { Headers } = require_headers();
22542
- var brandChecks = webidl.brandCheckMultiple([Headers, globalThis.Headers].filter(Boolean));
22541
+ var { Headers: Headers2 } = require_headers();
22542
+ var brandChecks = webidl.brandCheckMultiple([Headers2, globalThis.Headers].filter(Boolean));
22543
22543
  function getCookies(headers) {
22544
22544
  webidl.argumentLengthCheck(arguments, 1, "getCookies");
22545
22545
  brandChecks(headers);
@@ -23280,7 +23280,7 @@ var require_connection = __commonJS({
23280
23280
  var { parseExtensions, isClosed, isClosing, isEstablished, isConnecting, validateCloseCodeAndReason } = require_util5();
23281
23281
  var { makeRequest } = require_request2();
23282
23282
  var { fetching } = require_fetch();
23283
- var { Headers, getHeadersList } = require_headers();
23283
+ var { Headers: Headers2, getHeadersList } = require_headers();
23284
23284
  var { getDecodeSplit } = require_util2();
23285
23285
  var { WebsocketFrameSend } = require_frame();
23286
23286
  var assert = __require("assert");
@@ -23302,7 +23302,7 @@ var require_connection = __commonJS({
23302
23302
  useURLCredentials: true
23303
23303
  });
23304
23304
  if (options.headers) {
23305
- const headersList = getHeadersList(new Headers(options.headers));
23305
+ const headersList = getHeadersList(new Headers2(options.headers));
23306
23306
  request.headersList = headersList;
23307
23307
  }
23308
23308
  const keyValue = crypto2.randomBytes(16).toString("base64");
@@ -25411,10 +25411,10 @@ var require_undici = __commonJS({
25411
25411
  "use strict";
25412
25412
  var Client = require_client();
25413
25413
  var Dispatcher = require_dispatcher();
25414
- var Pool = require_pool();
25414
+ var Pool2 = require_pool();
25415
25415
  var BalancedPool = require_balanced_pool();
25416
25416
  var RoundRobinPool = require_round_robin_pool();
25417
- var Agent = require_agent();
25417
+ var Agent2 = require_agent();
25418
25418
  var ProxyAgent2 = require_proxy_agent();
25419
25419
  var Socks5ProxyAgent = require_socks5_proxy_agent();
25420
25420
  var EnvHttpProxyAgent = require_env_http_proxy_agent();
@@ -25438,10 +25438,10 @@ var require_undici = __commonJS({
25438
25438
  Object.assign(Dispatcher.prototype, api);
25439
25439
  module.exports.Dispatcher = Dispatcher;
25440
25440
  module.exports.Client = Client;
25441
- module.exports.Pool = Pool;
25441
+ module.exports.Pool = Pool2;
25442
25442
  module.exports.BalancedPool = BalancedPool;
25443
25443
  module.exports.RoundRobinPool = RoundRobinPool;
25444
- module.exports.Agent = Agent;
25444
+ module.exports.Agent = Agent2;
25445
25445
  module.exports.ProxyAgent = ProxyAgent2;
25446
25446
  module.exports.Socks5ProxyAgent = Socks5ProxyAgent;
25447
25447
  module.exports.EnvHttpProxyAgent = EnvHttpProxyAgent;
@@ -47245,7 +47245,7 @@ function closeLogger() {
47245
47245
  }
47246
47246
 
47247
47247
  // src/upstream-proxy.ts
47248
- var import_undici = __toESM(require_undici(), 1);
47248
+ var import_undici2 = __toESM(require_undici(), 1);
47249
47249
  import { execFileSync } from "child_process";
47250
47250
  import net from "net";
47251
47251
  import tls from "tls";
@@ -47323,6 +47323,179 @@ function maskHostInText(text, host) {
47323
47323
  return text;
47324
47324
  }
47325
47325
 
47326
+ // src/fetch-util.ts
47327
+ var import_undici = __toESM(require_undici(), 1);
47328
+ var MAX_REQUEST_BYTES = 100 * 1024 * 1024;
47329
+ var UPSTREAM_TIMEOUT_MS = 12 * 60 * 1e3;
47330
+ var liveUpstreamTimers = /* @__PURE__ */ new Set();
47331
+ function upstreamTimeoutMs() {
47332
+ const raw = Number(process.env.BILI_UPSTREAM_TIMEOUT_MS);
47333
+ return Number.isInteger(raw) && raw > 0 ? raw : UPSTREAM_TIMEOUT_MS;
47334
+ }
47335
+ var directDispatchers = /* @__PURE__ */ new Map();
47336
+ function directDispatcher(timeoutMs) {
47337
+ let agent = directDispatchers.get(timeoutMs);
47338
+ if (!agent) {
47339
+ agent = new import_undici.Agent({ headersTimeout: timeoutMs, bodyTimeout: timeoutMs });
47340
+ directDispatchers.set(timeoutMs, agent);
47341
+ }
47342
+ return agent;
47343
+ }
47344
+ async function fetchWithTimeout(url, opts, timeoutMs, externalSignal) {
47345
+ const effective = timeoutMs ?? upstreamTimeoutMs();
47346
+ const controller = new AbortController();
47347
+ let cleared = false;
47348
+ const armTimer = () => {
47349
+ const t = setTimeout(() => {
47350
+ liveUpstreamTimers.delete(t);
47351
+ controller.abort();
47352
+ }, effective);
47353
+ liveUpstreamTimers.add(t);
47354
+ return t;
47355
+ };
47356
+ let timer3 = armTimer();
47357
+ const rearm = () => {
47358
+ if (cleared) return;
47359
+ clearTimeout(timer3);
47360
+ liveUpstreamTimers.delete(timer3);
47361
+ timer3 = armTimer();
47362
+ };
47363
+ let onExternalAbort = null;
47364
+ if (externalSignal) {
47365
+ if (externalSignal.aborted) controller.abort();
47366
+ else {
47367
+ onExternalAbort = () => controller.abort();
47368
+ externalSignal.addEventListener("abort", onExternalAbort, { once: true });
47369
+ }
47370
+ }
47371
+ const cleanup = () => {
47372
+ cleared = true;
47373
+ clearTimeout(timer3);
47374
+ liveUpstreamTimers.delete(timer3);
47375
+ if (onExternalAbort && externalSignal) externalSignal.removeEventListener("abort", onExternalAbort);
47376
+ };
47377
+ try {
47378
+ const finalOpts = {
47379
+ ...opts,
47380
+ signal: controller.signal,
47381
+ dispatcher: opts.dispatcher ?? directDispatcher(effective)
47382
+ };
47383
+ const raw = await fetch(url, finalOpts);
47384
+ if (raw.body) {
47385
+ const wrapped = armIdleBody(raw.body, rearm);
47386
+ return {
47387
+ response: new Response(wrapped, {
47388
+ status: raw.status,
47389
+ statusText: raw.statusText,
47390
+ headers: raw.headers
47391
+ }),
47392
+ clearTimer: cleanup
47393
+ };
47394
+ }
47395
+ return { response: raw, clearTimer: cleanup };
47396
+ } catch (e) {
47397
+ cleanup();
47398
+ throw e;
47399
+ }
47400
+ }
47401
+ function armIdleBody(body, rearm) {
47402
+ const reader = body.getReader();
47403
+ return new ReadableStream({
47404
+ async pull(controller) {
47405
+ try {
47406
+ const result = await reader.read();
47407
+ if (result.done) {
47408
+ controller.close();
47409
+ return;
47410
+ }
47411
+ rearm();
47412
+ controller.enqueue(result.value);
47413
+ } catch (e) {
47414
+ controller.error(e);
47415
+ }
47416
+ },
47417
+ async cancel(reason) {
47418
+ try {
47419
+ await reader.cancel(reason);
47420
+ } catch {
47421
+ }
47422
+ }
47423
+ });
47424
+ }
47425
+ var UpstreamHttpError = class extends Error {
47426
+ status;
47427
+ body;
47428
+ attempts;
47429
+ constructor(status, body, attempts) {
47430
+ super(`upstream error ${status}`);
47431
+ this.name = "UpstreamHttpError";
47432
+ this.status = status;
47433
+ this.body = body;
47434
+ this.attempts = attempts;
47435
+ }
47436
+ };
47437
+ var TRANSIENT_BODY_MARKERS = [
47438
+ "captcha",
47439
+ "verify failed",
47440
+ "risk control",
47441
+ "\u98CE\u63A7",
47442
+ "rate limit",
47443
+ "too many requests",
47444
+ "try again"
47445
+ ];
47446
+ function isTransientUpstreamError(status, body) {
47447
+ if (status === 429 || status >= 500) return true;
47448
+ if (status < 400) return false;
47449
+ const lower = body.toLowerCase();
47450
+ return TRANSIENT_BODY_MARKERS.some((marker) => lower.includes(marker));
47451
+ }
47452
+ var REPLAY_MAX_ATTEMPTS = 3;
47453
+ function replayMaxAttempts() {
47454
+ const raw = Number(process.env.BILI_REPLAY_RETRY_MAX);
47455
+ return Number.isInteger(raw) && raw >= 1 ? raw : REPLAY_MAX_ATTEMPTS;
47456
+ }
47457
+ function replayBaseDelayMs() {
47458
+ const raw = Number(process.env.BILI_REPLAY_RETRY_BASE_MS);
47459
+ return Number.isFinite(raw) && raw >= 0 ? raw : 1500;
47460
+ }
47461
+ function maxShrinkPerCompress() {
47462
+ const raw = Number(process.env.BILI_MAX_SHRINK_PER_COMPRESS);
47463
+ return Number.isFinite(raw) && raw > 0 && raw <= 1 ? raw : void 0;
47464
+ }
47465
+ function replayBackoffMs(attempt) {
47466
+ return replayBaseDelayMs() * 2 ** (attempt - 1);
47467
+ }
47468
+ function sleep(ms2, signal) {
47469
+ if (ms2 <= 0 || signal?.aborted) return Promise.resolve();
47470
+ return new Promise((resolve) => {
47471
+ let timer3 = null;
47472
+ const finish2 = () => {
47473
+ if (timer3) clearTimeout(timer3);
47474
+ if (signal) signal.removeEventListener("abort", finish2);
47475
+ resolve();
47476
+ };
47477
+ timer3 = setTimeout(finish2, ms2);
47478
+ if (signal) signal.addEventListener("abort", finish2, { once: true });
47479
+ });
47480
+ }
47481
+ async function fetchWithRetry(url, opts, timeoutMs, externalSignal, onRetry) {
47482
+ const maxAttempts = replayMaxAttempts();
47483
+ for (let attempt = 1; ; attempt++) {
47484
+ const result = await fetchWithTimeout(url, opts, timeoutMs, externalSignal);
47485
+ if (result.response.ok) return result;
47486
+ const errText2 = await result.response.text().catch(() => "upstream error");
47487
+ result.clearTimer();
47488
+ const lastAttempt = attempt >= maxAttempts;
47489
+ if (!lastAttempt && isTransientUpstreamError(result.response.status, errText2)) {
47490
+ const delayMs = replayBackoffMs(attempt);
47491
+ onRetry?.({ attempt, status: result.response.status, detail: errText2, delayMs, maxAttempts });
47492
+ await sleep(delayMs, externalSignal);
47493
+ continue;
47494
+ }
47495
+ throw new UpstreamHttpError(result.response.status, errText2, attempt);
47496
+ }
47497
+ }
47498
+
47326
47499
  // src/upstream-proxy.ts
47327
47500
  var dispatcherCache = /* @__PURE__ */ new Map();
47328
47501
  var lastConnection = {};
@@ -47519,12 +47692,21 @@ function resolveProxyDecision(routes, globalProxy, upstreamUrl, fallback = {}) {
47519
47692
  function resolveProxy(routes, globalProxy, upstreamUrl, fallback = {}) {
47520
47693
  return resolveProxyDecision(routes, globalProxy, upstreamUrl, fallback).proxy;
47521
47694
  }
47522
- function proxyDispatcher(proxyUrl) {
47695
+ function proxyDispatcher(proxyUrl, timeoutMs) {
47523
47696
  if (!proxyUrl) return void 0;
47524
- let agent = dispatcherCache.get(proxyUrl);
47697
+ const t = timeoutMs ?? upstreamTimeoutMs();
47698
+ const key = `${proxyUrl}\0${t}`;
47699
+ let agent = dispatcherCache.get(key);
47525
47700
  if (!agent) {
47526
- agent = new import_undici.ProxyAgent({ uri: proxyUrl });
47527
- dispatcherCache.set(proxyUrl, agent);
47701
+ const withTimeouts = (options) => ({ ...options, headersTimeout: t, bodyTimeout: t });
47702
+ agent = new import_undici2.ProxyAgent({
47703
+ uri: proxyUrl,
47704
+ headersTimeout: t,
47705
+ bodyTimeout: t,
47706
+ factory: (origin, options) => new import_undici2.Pool(origin, withTimeouts(options)),
47707
+ clientFactory: (origin, options) => new import_undici2.Pool(origin, withTimeouts(options))
47708
+ });
47709
+ dispatcherCache.set(key, agent);
47528
47710
  }
47529
47711
  return agent;
47530
47712
  }
@@ -47677,6 +47859,95 @@ function getUpstreamConnectionStatus() {
47677
47859
  return { ...lastConnection };
47678
47860
  }
47679
47861
 
47862
+ // src/compat-roles.ts
47863
+ function parseCompatRoles(v2) {
47864
+ if (!v2 || typeof v2 !== "object" || Array.isArray(v2)) return void 0;
47865
+ const obj = v2;
47866
+ let out;
47867
+ for (const [k2, val] of Object.entries(obj)) {
47868
+ if (typeof val !== "string" || val.length === 0) continue;
47869
+ out ??= {};
47870
+ out[k2] = val;
47871
+ }
47872
+ return out;
47873
+ }
47874
+ function resolveCompatRoles(routes, upstreamUrl, globalRoles) {
47875
+ const providerRoles = findRoute(routes, upstreamUrl)?.compat?.roles;
47876
+ if (!globalRoles && !providerRoles) return {};
47877
+ return { ...globalRoles, ...providerRoles };
47878
+ }
47879
+ function applyCompatRolesJson(parsed, protocol, roles) {
47880
+ if (Object.keys(roles).length === 0) return 0;
47881
+ const items = protocol === "openai" ? parsed.messages : parsed.input;
47882
+ if (!Array.isArray(items)) return 0;
47883
+ let rewritten = 0;
47884
+ for (const item of items) {
47885
+ if (!item || typeof item !== "object" || Array.isArray(item)) continue;
47886
+ const msg = item;
47887
+ if (protocol === "responses" && msg.type !== void 0 && msg.type !== "message") continue;
47888
+ const role = msg.role;
47889
+ if (typeof role !== "string") continue;
47890
+ const mapped = roles[role];
47891
+ if (mapped === void 0 || mapped === role) continue;
47892
+ msg.role = mapped;
47893
+ rewritten++;
47894
+ }
47895
+ return rewritten;
47896
+ }
47897
+ var ROLE_CAPTURE_STOPWORDS = /* @__PURE__ */ new Set([
47898
+ "must",
47899
+ "should",
47900
+ "be",
47901
+ "is",
47902
+ "one",
47903
+ "of",
47904
+ "the",
47905
+ "a",
47906
+ "an",
47907
+ "in",
47908
+ "for",
47909
+ "not",
47910
+ "was",
47911
+ "and",
47912
+ "or",
47913
+ "to",
47914
+ "only",
47915
+ "allowed",
47916
+ "supported",
47917
+ "valid",
47918
+ "value",
47919
+ "message",
47920
+ "messages"
47921
+ ]);
47922
+ var ROLE_REJECTION_PATTERNS = [
47923
+ /(?:invalid|unknown|unsupported|unrecognized|unexpected)[a-z ]{0,24}?role\s*[:=]?\s*["'`]?([a-z][a-z0-9_-]{1,31})["'`]?/i,
47924
+ /role\s*[:=]?\s*["'`]?([a-z][a-z0-9_-]{1,31})["'`]?\s+(?:is\s+)?not\s+(?:supported|allowed|recognized|valid|accepted)/i
47925
+ ];
47926
+ function detectRoleRejection(status, bodyText) {
47927
+ if (status !== 400) return null;
47928
+ const head = bodyText.slice(0, 4096);
47929
+ for (const re2 of ROLE_REJECTION_PATTERNS) {
47930
+ const m2 = re2.exec(head);
47931
+ if (!m2) continue;
47932
+ const role = m2[1].toLowerCase();
47933
+ if (ROLE_CAPTURE_STOPWORDS.has(role)) continue;
47934
+ return { role };
47935
+ }
47936
+ return null;
47937
+ }
47938
+ function applyCompatRoles(body, protocol, roles) {
47939
+ if (Object.keys(roles).length === 0) return { body, rewritten: 0 };
47940
+ let parsed;
47941
+ try {
47942
+ parsed = JSON.parse(body);
47943
+ } catch {
47944
+ return { body, rewritten: 0 };
47945
+ }
47946
+ const rewritten = applyCompatRolesJson(parsed, protocol, roles);
47947
+ if (rewritten === 0) return { body, rewritten: 0 };
47948
+ return { body: JSON.stringify(parsed), rewritten };
47949
+ }
47950
+
47680
47951
  // src/config.ts
47681
47952
  function safeReadJson(path18) {
47682
47953
  try {
@@ -47823,6 +48094,7 @@ function loadOptions(env = process.env) {
47823
48094
  promptCache: {
47824
48095
  routing: parsePromptCacheRouting(env.ACP_PROMPT_CACHE_ROUTING ?? fileConfig.promptCache?.routing)
47825
48096
  },
48097
+ compat: { roles: parseCompatRoles(fileConfig.compat?.roles) ?? {} },
47826
48098
  sessionHeader: env.ACP_SESSION_HEADER ?? fileConfig.sessionHeader ?? "x-acp-session",
47827
48099
  log: env.ACP_LOG !== "0" && fileConfig.log !== false,
47828
48100
  debug: (env.ACP_DEBUG ?? (fileConfig.debug ? "1" : "0")) === "1",
@@ -47892,6 +48164,8 @@ function parseRouteEntry(v2) {
47892
48164
  if (typeof obj.proxy === "string") route.proxy = obj.proxy;
47893
48165
  if (obj.compressProtocol === "marker" || obj.compressProtocol === "tools") route.compressProtocol = obj.compressProtocol;
47894
48166
  if (obj.compress) route.compress = obj.compress;
48167
+ const compatRoles = parseCompatRoles(obj.compat?.roles);
48168
+ if (compatRoles) route.compat = { roles: compatRoles };
47895
48169
  return route;
47896
48170
  }
47897
48171
  if (v2 === null) return {};
@@ -48072,157 +48346,6 @@ import { readFile, writeFile, mkdir } from "fs/promises";
48072
48346
  import { existsSync as existsSync2, statSync as statSync2 } from "fs";
48073
48347
  import path3 from "path";
48074
48348
 
48075
- // src/fetch-util.ts
48076
- var MAX_REQUEST_BYTES = 100 * 1024 * 1024;
48077
- var UPSTREAM_TIMEOUT_MS = 10 * 60 * 1e3;
48078
- var liveUpstreamTimers = /* @__PURE__ */ new Set();
48079
- async function fetchWithTimeout(url, opts, timeoutMs = UPSTREAM_TIMEOUT_MS, externalSignal) {
48080
- const controller = new AbortController();
48081
- const armTimer = () => {
48082
- const t = setTimeout(() => {
48083
- liveUpstreamTimers.delete(t);
48084
- controller.abort();
48085
- }, timeoutMs);
48086
- liveUpstreamTimers.add(t);
48087
- return t;
48088
- };
48089
- let timer3 = armTimer();
48090
- const rearm = () => {
48091
- clearTimeout(timer3);
48092
- liveUpstreamTimers.delete(timer3);
48093
- timer3 = armTimer();
48094
- };
48095
- let onExternalAbort = null;
48096
- if (externalSignal) {
48097
- if (externalSignal.aborted) controller.abort();
48098
- else {
48099
- onExternalAbort = () => controller.abort();
48100
- externalSignal.addEventListener("abort", onExternalAbort, { once: true });
48101
- }
48102
- }
48103
- const cleanup = () => {
48104
- clearTimeout(timer3);
48105
- liveUpstreamTimers.delete(timer3);
48106
- if (onExternalAbort && externalSignal) externalSignal.removeEventListener("abort", onExternalAbort);
48107
- };
48108
- try {
48109
- const finalOpts = { ...opts, signal: controller.signal };
48110
- const raw = await fetch(url, finalOpts);
48111
- if (raw.body) {
48112
- const wrapped = armIdleBody(raw.body, rearm);
48113
- return {
48114
- response: new Response(wrapped, {
48115
- status: raw.status,
48116
- statusText: raw.statusText,
48117
- headers: raw.headers
48118
- }),
48119
- clearTimer: cleanup
48120
- };
48121
- }
48122
- return { response: raw, clearTimer: cleanup };
48123
- } catch (e) {
48124
- cleanup();
48125
- throw e;
48126
- }
48127
- }
48128
- function armIdleBody(body, rearm) {
48129
- const reader = body.getReader();
48130
- return new ReadableStream({
48131
- async pull(controller) {
48132
- try {
48133
- const result = await reader.read();
48134
- if (result.done) {
48135
- controller.close();
48136
- return;
48137
- }
48138
- rearm();
48139
- controller.enqueue(result.value);
48140
- } catch (e) {
48141
- controller.error(e);
48142
- }
48143
- },
48144
- async cancel(reason) {
48145
- try {
48146
- await reader.cancel(reason);
48147
- } catch {
48148
- }
48149
- }
48150
- });
48151
- }
48152
- var UpstreamHttpError = class extends Error {
48153
- status;
48154
- body;
48155
- attempts;
48156
- constructor(status, body, attempts) {
48157
- super(`upstream error ${status}`);
48158
- this.name = "UpstreamHttpError";
48159
- this.status = status;
48160
- this.body = body;
48161
- this.attempts = attempts;
48162
- }
48163
- };
48164
- var TRANSIENT_BODY_MARKERS = [
48165
- "captcha",
48166
- "verify failed",
48167
- "risk control",
48168
- "\u98CE\u63A7",
48169
- "rate limit",
48170
- "too many requests",
48171
- "try again"
48172
- ];
48173
- function isTransientUpstreamError(status, body) {
48174
- if (status === 429 || status >= 500) return true;
48175
- if (status < 400) return false;
48176
- const lower = body.toLowerCase();
48177
- return TRANSIENT_BODY_MARKERS.some((marker) => lower.includes(marker));
48178
- }
48179
- var REPLAY_MAX_ATTEMPTS = 3;
48180
- function replayMaxAttempts() {
48181
- const raw = Number(process.env.BILI_REPLAY_RETRY_MAX);
48182
- return Number.isInteger(raw) && raw >= 1 ? raw : REPLAY_MAX_ATTEMPTS;
48183
- }
48184
- function replayBaseDelayMs() {
48185
- const raw = Number(process.env.BILI_REPLAY_RETRY_BASE_MS);
48186
- return Number.isFinite(raw) && raw >= 0 ? raw : 1500;
48187
- }
48188
- function maxShrinkPerCompress() {
48189
- const raw = Number(process.env.BILI_MAX_SHRINK_PER_COMPRESS);
48190
- return Number.isFinite(raw) && raw > 0 && raw <= 1 ? raw : void 0;
48191
- }
48192
- function replayBackoffMs(attempt) {
48193
- return replayBaseDelayMs() * 2 ** (attempt - 1);
48194
- }
48195
- function sleep(ms2, signal) {
48196
- if (ms2 <= 0 || signal?.aborted) return Promise.resolve();
48197
- return new Promise((resolve) => {
48198
- let timer3 = null;
48199
- const finish2 = () => {
48200
- if (timer3) clearTimeout(timer3);
48201
- if (signal) signal.removeEventListener("abort", finish2);
48202
- resolve();
48203
- };
48204
- timer3 = setTimeout(finish2, ms2);
48205
- if (signal) signal.addEventListener("abort", finish2, { once: true });
48206
- });
48207
- }
48208
- async function fetchWithRetry(url, opts, timeoutMs, externalSignal, onRetry) {
48209
- const maxAttempts = replayMaxAttempts();
48210
- for (let attempt = 1; ; attempt++) {
48211
- const result = await fetchWithTimeout(url, opts, timeoutMs, externalSignal);
48212
- if (result.response.ok) return result;
48213
- const errText2 = await result.response.text().catch(() => "upstream error");
48214
- result.clearTimer();
48215
- const lastAttempt = attempt >= maxAttempts;
48216
- if (!lastAttempt && isTransientUpstreamError(result.response.status, errText2)) {
48217
- const delayMs = replayBackoffMs(attempt);
48218
- onRetry?.({ attempt, status: result.response.status, detail: errText2, delayMs, maxAttempts });
48219
- await sleep(delayMs, externalSignal);
48220
- continue;
48221
- }
48222
- throw new UpstreamHttpError(result.response.status, errText2, attempt);
48223
- }
48224
- }
48225
-
48226
48349
  // src/registry-snapshot.json
48227
48350
  var registry_snapshot_default = { fetchedAt: "2026-08-24T10:39:50.602Z", count: 355, models: { "tencent/hy3-preview": { id: "tencent/hy3-preview", name: "Hy3 preview", description: "Tencent Hy reasoning model for coding, instruction following, and agent tasks", family: "Hy", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-04-20", last_updated: "2026-04-20", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 64e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/tencent/Hy3-preview" }], benchmarks: [{ name: "SWE-Bench Verified", score: 74.4, metric: "resolved", source: "https://huggingface.co/tencent/Hy3-preview" }] }, "tencent/hy3": { id: "tencent/hy3", name: "Hy3", description: "Tencent Hy reasoning model for coding, instruction following, and agent tasks", family: "Hy", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-07-06", last_updated: "2026-07-06", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 64e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/tencent/Hy3" }], benchmarks: [{ name: "SWE-Bench Verified", score: 78, metric: "resolved", source: "https://huggingface.co/tencent/Hy3" }] }, "nvidia/llama-3.3-nemotron-super-49b-v1": { id: "nvidia/llama-3.3-nemotron-super-49b-v1", name: "Llama 3.3 Nemotron Super 49B v1", description: "Nemotron model for efficient reasoning, coding, and specialized AI agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-04-07", last_updated: "2025-04-07", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 } }, "nvidia/nemotron-3-nano-30b-a3b": { id: "nvidia/nemotron-3-nano-30b-a3b", name: "Nemotron 3 Nano 30B A3B", description: "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-12-15", last_updated: "2025-12-15", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 } }, "nvidia/nemotron-3.5-content-safety": { id: "nvidia/nemotron-3.5-content-safety", name: "Nemotron 3.5 Content Safety", description: "Safety model for policy screening, moderation, and risk-aware routing workflows", family: "nemotron", attachment: true, reasoning: true, tool_call: false, temperature: true, release_date: "2026-06-04", last_updated: "2026-06-04", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/nemotron-cascade-2-30b-a3b": { id: "nvidia/nemotron-cascade-2-30b-a3b", name: "Nemotron Cascade 2 30B A3B", description: "Nemotron model for efficient reasoning, coding, and specialized AI agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-03-24", last_updated: "2026-04-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 32768 } }, "nvidia/nemotron-3-content-safety": { id: "nvidia/nemotron-3-content-safety", name: "Nemotron 3 Content Safety", description: "Safety model for policy screening, moderation, and risk-aware routing workflows", family: "nemotron", attachment: false, reasoning: false, tool_call: false, temperature: false, release_date: "2026-04-16", last_updated: "2026-04-16", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 } }, "nvidia/llama-3.1-nemotron-70b-instruct": { id: "nvidia/llama-3.1-nemotron-70b-instruct", name: "Llama 3.1 Nemotron 70B Instruct", description: "Nemotron model for efficient reasoning, coding, and specialized AI agents", family: "nemotron", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-04-15", last_updated: "2025-04-15", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/nemotron-nano-12b-v2-vl": { id: "nvidia/nemotron-nano-12b-v2-vl", name: "Nemotron Nano 12B v2 VL", description: "Nemotron multimodal model for visual reasoning and agentic AI workflows", family: "nemotron", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2025-10-28", last_updated: "2025-10-28", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 } }, "nvidia/nemotron-voicechat": { id: "nvidia/nemotron-voicechat", name: "Nemotron VoiceChat", description: "Nemotron multimodal model for visual reasoning and agentic AI workflows", family: "nemotron", attachment: true, reasoning: false, tool_call: true, temperature: true, release_date: "2026-03-16", last_updated: "2026-03-16", modalities: { input: ["text", "audio"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/llama-nemotron-rerank-vl-1b-v2": { id: "nvidia/llama-nemotron-rerank-vl-1b-v2", name: "Llama Nemotron Rerank VL 1B v2", description: "Reranking model for improving retrieval quality in search and recommendation systems", family: "nemotron", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-03-31", last_updated: "2026-03-31", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 } }, "nvidia/llama-3.3-nemotron-super-49b-v1.5": { id: "nvidia/llama-3.3-nemotron-super-49b-v1.5", name: "Llama 3.3 Nemotron Super 49B v1.5", description: "Nemotron model for efficient reasoning, coding, and specialized AI agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-07-25", last_updated: "2025-07-25", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 } }, "nvidia/nemotron-3-super-120b-a12b": { id: "nvidia/nemotron-3-super-120b-a12b", name: "Nemotron 3 Super 120B A12B", description: "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-03-11", last_updated: "2026-03-11", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 } }, "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": { id: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", name: "Nemotron 3 Nano Omni 30B A3B Reasoning", description: "Open Nemotron omni model combining reasoning with text, vision, and audio", family: "nemotron", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-04-28", last_updated: "2026-04-28", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 65536 } }, "nvidia/nemotron-3.5-lightning": { id: "nvidia/nemotron-3.5-lightning", name: "Nemotron 3.5 Lightning 30B A3B", description: "Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads", family: "nemotron", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-11", last_updated: "2026-08-11", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 } }, "nvidia/nemotron-content-safety-reasoning-4b": { id: "nvidia/nemotron-content-safety-reasoning-4b", name: "Nemotron Content Safety Reasoning 4B", description: "Safety model for policy screening, moderation, and risk-aware routing workflows", family: "nemotron", attachment: false, reasoning: true, tool_call: false, temperature: false, release_date: "2026-01-22", last_updated: "2026-01-22", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 } }, "nvidia/nemotron-3-ultra-550b-a55b": { id: "nvidia/nemotron-3-ultra-550b-a55b", name: "Nemotron 3 Ultra 550B A55B", description: "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-04", last_updated: "2026-06-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Verified", score: 70.7, metric: "resolved", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "SWE-Bench Multilingual", score: 67.7, metric: "resolve rate", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "Terminal-Bench", score: 56.4, metric: "success rate", version: "2.1", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "GPQA", score: 87, metric: "accuracy", variant: "no tools", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "Humanity's Last Exam", score: 26.7, metric: "accuracy", variant: "no tools", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "Humanity's Last Exam", score: 37.4, metric: "accuracy", variant: "with tools", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "LiveCodeBench", score: 89, metric: "pass@1", version: "v6", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "MMLU-Pro", score: 86.8, metric: "accuracy", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "BrowseComp", score: 44.4, metric: "accuracy", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "IFBench", score: 81.7, metric: "accuracy", variant: "prompt loose", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "GDPval", score: 46.7, metric: "wins or ties", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }] }, "nvidia/mistral-nemotron": { id: "nvidia/mistral-nemotron", name: "Mistral Nemotron", description: "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", family: "nemotron", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-06-11", last_updated: "2025-06-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/llama-3.1-nemotron-safety-guard-8b-v3": { id: "nvidia/llama-3.1-nemotron-safety-guard-8b-v3", name: "Llama 3.1 Nemotron Safety Guard 8B v3", description: "Safety model for policy screening, moderation, and risk-aware routing workflows", family: "nemotron", attachment: false, reasoning: false, tool_call: false, temperature: false, release_date: "2025-10-28", last_updated: "2025-10-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 } }, "nvidia/nemotron-nano-9b-v2": { id: "nvidia/nemotron-nano-9b-v2", name: "Nemotron Nano 9B v2", description: "Compact Nemotron model for efficient reasoning and deployable AI agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-08-18", last_updated: "2025-08-18", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 } }, "nvidia/llama-3.1-nemotron-ultra-253b": { id: "nvidia/llama-3.1-nemotron-ultra-253b", name: "Llama 3.1 Nemotron Ultra 253B", description: "Flagship Nemotron model for high-throughput reasoning and complex agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-04-07", last_updated: "2025-04-07", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/nemotron-mini-4b-instruct": { id: "nvidia/nemotron-mini-4b-instruct", name: "Nemotron Mini 4B Instruct", description: "Compact Nemotron model for efficient reasoning and deployable AI agents", family: "nemotron", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2024-08-21", last_updated: "2024-08-26", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/llama-nemotron-embed-vl-1b-v2": { id: "nvidia/llama-nemotron-embed-vl-1b-v2", name: "Llama Nemotron Embed VL 1B v2", description: "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", family: "nemotron", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-02-10", last_updated: "2026-02-10", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 32768, output: 2048 } }, "aisingapore/gemma-sea-lion-v4-27b-it": { id: "aisingapore/gemma-sea-lion-v4-27b-it", name: "Gemma-SEA-LION-v4-27B-IT", description: "Gemma 3 27B tuned by AI Singapore for Southeast Asian languages and instruction following", family: "gemma", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2025-09-23", last_updated: "2025-09-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4-27B-IT" }] }, "microsoft/phi-4-mini": { id: "microsoft/phi-4-mini", name: "Phi-4-mini", description: "Compact Microsoft instruction model tuned for efficient coding assistance, reasoning, and low-latency agent tasks", family: "phi", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2023-10", release_date: "2024-12-11", last_updated: "2024-12-11", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 }, links: [{ label: "Weights", url: "https://huggingface.co/microsoft/Phi-4-mini-instruct", type: "weights" }], benchmarks: [{ name: "MMLU", score: 67.3, metric: "accuracy", source: "https://huggingface.co/microsoft/Phi-4-mini-instruct/resolve/main/README.md" }] }, "microsoft/mai-code-1-flash": { id: "microsoft/mai-code-1-flash", name: "MAI-Code-1-Flash", description: "Microsoft coding model built for fast, efficient assistance in everyday developer workflows", family: "mai", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-12", release_date: "2026-06-02", last_updated: "2026-06-08", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 128e3 }, links: [{ label: "Model card", url: "https://microsoft.ai/pdf/MAI-Code-1-Flash-Model-Card.PDF", type: "model_card" }, { label: "Announcement", url: "https://microsoft.ai/news/introducingmai-code-1-flash/", type: "announcement" }], benchmarks: [{ name: "SWE-Bench Pro", score: 51.2, metric: "resolve rate", harness: "GitHub Copilot", source: "https://microsoft.ai/news/introducingmai-code-1-flash/", date: "2026-06-02" }, { name: "SWE-Bench Verified", score: 71.6, metric: "resolved", source: "https://llm-stats.com/benchmarks/swe-bench-verified" }, { name: "Terminal-Bench", score: 54.8, metric: "success rate", version: "2.0", source: "https://llm-stats.com/benchmarks/terminal-bench-2" }, { name: "GPQA Diamond", score: 84.6, metric: "accuracy", source: "https://llm-stats.com/benchmarks/gpqa" }] }, "microsoft/mai-code-1.1-flash": { id: "microsoft/mai-code-1.1-flash", name: "MAI-Code-1.1-Flash", description: "Microsoft coding model with native vision support, optimized for fast and efficient software development", family: "mai", attachment: true, reasoning: true, tool_call: true, structured_output: true, release_date: "2026-08-11", last_updated: "2026-08-11", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 128e3 }, links: [{ label: "Announcement", url: "https://microsoft.ai/news/mai-code-1-1-flash-br-better-faster-at-a-quarter-of-the-cost/", type: "announcement" }] }, "deepseek/deepseek-r1-distill-qwen-32b": { id: "deepseek/deepseek-r1-distill-qwen-32b", name: "DeepSeek-R1-Distill-Qwen-32B", description: "R1 reasoning distilled into Qwen 2.5 32B for efficient open-weight step-by-step problem solving", family: "deepseek-thinking", attachment: false, reasoning: true, tool_call: false, temperature: true, release_date: "2025-01-20", last_updated: "2025-01-20", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B" }] }, "deepseek/deepseek-r1": { id: "deepseek/deepseek-r1", name: "DeepSeek-R1", description: "Classic open reasoning model for transparent math, coding, and deliberate problem solving", family: "deepseek-thinking", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-07", release_date: "2025-01-20", last_updated: "2025-05-29", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-R1" }], benchmarks: [{ name: "Aider Polyglot", score: 56.9, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-01-20" }, { name: "Artificial Analysis Coding Index", score: 15.9, metric: "index", source: "https://openrouter.ai/deepseek/deepseek-r1/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 35.7, metric: "percent correct", source: "https://openrouter.ai/deepseek/deepseek-r1/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 6.1, metric: "success rate", source: "https://openrouter.ai/deepseek/deepseek-r1/benchmarks", date: "2026-03-11" }] }, "deepseek/deepseek-v3-0324": { id: "deepseek/deepseek-v3-0324", name: "DeepSeek V3 0324", description: "March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding", family: "deepseek", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-03-24", last_updated: "2025-03-24", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 163840, output: 163840 }, weights: [{ label: "Model weights", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3-0324", format: "safetensors" }] }, "deepseek/deepseek-v4-flash": { id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash", description: "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", family: "deepseek-flash", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-05", release_date: "2026-04-24", last_updated: "2026-04-24", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash" }], benchmarks: [{ name: "SWE-Bench Verified", score: 79, metric: "resolved", source: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash" }] }, "deepseek/deepseek-v4-pro-0813": { id: "deepseek/deepseek-v4-pro-0813", name: "DeepSeek V4 Pro 0813", description: "DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes", family: "deepseek-thinking", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-12", last_updated: "2026-08-22", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, license: "MIT", weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813" }] }, "deepseek/deepseek-v3.1": { id: "deepseek/deepseek-v3.1", name: "DeepSeek-V3.1", description: "Hybrid-reasoning DeepSeek model with thinking and non-thinking modes", family: "deepseek", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-08-21", last_updated: "2025-08-21", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "MIT License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3.1" }] }, "deepseek/deepseek-v4-pro": { id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro", description: "Open MoE flagship with million-token context for coding and long agent runs", family: "deepseek-thinking", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-05", release_date: "2026-04-24", last_updated: "2026-04-24", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro" }], benchmarks: [{ name: "SWE-Bench Verified", score: 80.6, metric: "resolved", source: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro" }, { name: "Artificial Analysis Coding Agent Index", score: 50.1, metric: "average pass@1", harness: "Claude Code", variant: "high", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 67.8, metric: "pass@1", harness: "Claude Code", variant: "high", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 18, metric: "pass@1", harness: "Claude Code", variant: "high", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 64.7, metric: "pass@1", harness: "Claude Code", variant: "high", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }] }, "deepseek/deepseek-v3": { id: "deepseek/deepseek-v3", name: "DeepSeek-V3", description: "Open DeepSeek MoE chat model for coding, math, and general reasoning", family: "deepseek", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2024-12-26", last_updated: "2024-12-26", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "DeepSeek Model License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3" }] }, "deepseek/deepseek-v3.2": { id: "deepseek/deepseek-v3.2", name: "DeepSeek V3.2", description: "Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use", family: "deepseek", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-07", release_date: "2025-12-01", last_updated: "2025-12-01", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 64e3 }, license: "MIT License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3.2" }] }, "deepseek/deepseek-chat": { id: "deepseek/deepseek-chat", name: "DeepSeek Chat", description: "DeepSeek chat model for instruction following, coding, and analysis", family: "deepseek", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-09", release_date: "2025-12-01", last_updated: "2026-02-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3.2" }], benchmarks: [{ name: "Aider Polyglot", score: 70.2, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-10-03" }] }, "deepseek/deepseek-reasoner": { id: "deepseek/deepseek-reasoner", name: "DeepSeek Reasoner", description: "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", family: "deepseek-thinking", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-09", release_date: "2025-12-01", last_updated: "2026-02-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3.2" }], benchmarks: [{ name: "Aider Polyglot", score: 74.2, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-10-03" }] }, "deepseek/deepseek-ocr-2": { id: "deepseek/deepseek-ocr-2", name: "DeepSeek OCR 2", description: "High-accuracy OCR model for extracting text from documents, screenshots, receipts, and natural scenes", attachment: true, reasoning: false, tool_call: false, release_date: "2026-01-27", last_updated: "2026-01-27", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 8192, output: 8192 } }, "deepseek/deepseek-v4-flash-0731": { id: "deepseek/deepseek-v4-flash-0731", name: "DeepSeek V4 Flash 0731", description: "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding", family: "deepseek-flash", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-05", release_date: "2026-07-31", last_updated: "2026-07-31", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, license: "MIT", weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731" }], benchmarks: [{ name: "Terminal-Bench", score: 82.7, metric: "pass@1", variant: "max", version: "2.1", source: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731" }, { name: "NL2Repo", score: 54.2, metric: "resolve rate", harness: "DeepSeek Harness minimal mode", variant: "max effort", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "CyberGym", score: 76.7, metric: "score", harness: "DeepSeek Harness minimal mode", variant: "max effort", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "DeepSWE", score: 54.4, metric: "resolve rate", harness: "DeepSeek Harness minimal mode", variant: "max effort", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "Toolathlon-Verified", score: 70.3, metric: "score", harness: "DeepSeek Harness minimal mode", variant: "max effort", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "Agents' Last Exam", score: 25.2, metric: "score", harness: "DeepSeek Harness minimal mode", variant: "max effort", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "AutomationBench", score: 25.1, metric: "success rate", variant: "max effort", dataset: "public", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "DSBench-FullStack", score: 68.7, metric: "score", variant: "max effort", dataset: "internal", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "DSBench-Hard", score: 59.6, metric: "score", variant: "max effort", dataset: "internal", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }] }, "deepseek/deepseek-v4-pro-0423": { id: "deepseek/deepseek-v4-pro-0423", name: "DeepSeek V4 Pro 0423", description: "DeepSeek V4 Pro initial snapshot with million-token context and support for thinking and non-thinking modes", family: "deepseek-thinking", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-05", release_date: "2026-04-23", last_updated: "2026-04-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 } }, "deepseek/deepseek-v4-flash-vision-exp": { id: "deepseek/deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision Exp", description: "Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work", family: "deepseek-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-21", last_updated: "2026-08-21", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 384e3 } }, "arcee-ai/trinity-large-preview": { id: "arcee-ai/trinity-large-preview", name: "Trinity Large Preview", description: "Lightly post-trained 398B MoE chat model for creative work, long-context prompts, and tool-using agents", family: "trinity", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2026-01-27", last_updated: "2026-05-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 524288, output: 262144 }, license: "OpenMDW-1.1", links: [{ label: "Model card", url: "https://huggingface.co/arcee-ai/Trinity-Large-Preview", type: "model_card" }, { label: "Announcement", url: "https://www.arcee.ai/blog/trinity-large", type: "announcement" }, { label: "License", url: "https://huggingface.co/arcee-ai/Trinity-Large-Preview/blob/main/LICENSE", type: "license" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/arcee-ai/Trinity-Large-Preview", format: "safetensors" }] }, "arcee-ai/trinity-mini": { id: "arcee-ai/trinity-mini", name: "Trinity Mini", description: "Reasoning-tuned 26B MoE model with 3B active parameters for agents, tools, and multi-step workloads", family: "trinity", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-12-01", last_updated: "2026-05-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 }, license: "OpenMDW-1.1", links: [{ label: "Model card", url: "https://huggingface.co/arcee-ai/Trinity-Mini", type: "model_card" }, { label: "Announcement", url: "https://www.arcee.ai/blog/the-trinity-manifesto", type: "announcement" }, { label: "License", url: "https://huggingface.co/arcee-ai/Trinity-Mini/blob/main/LICENSE", type: "license" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/arcee-ai/Trinity-Mini", format: "safetensors" }] }, "arcee-ai/trinity-large-thinking": { id: "arcee-ai/trinity-large-thinking", name: "Trinity Large Thinking", description: "Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use", family: "trinity", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-04-01", last_updated: "2026-05-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 524288, output: 262144 }, license: "OpenMDW-1.1", links: [{ label: "Model card", url: "https://huggingface.co/arcee-ai/Trinity-Large-Thinking", type: "model_card" }, { label: "Announcement", url: "https://www.arcee.ai/blog/trinity-large-thinking", type: "announcement" }, { label: "License", url: "https://huggingface.co/arcee-ai/Trinity-Large-Thinking/blob/main/LICENSE", type: "license" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/arcee-ai/Trinity-Large-Thinking", format: "safetensors" }] }, "arcee-ai/trinity-nano-preview": { id: "arcee-ai/trinity-nano-preview", name: "Trinity Nano Preview", description: "Experimental chat-tuned 6B MoE model with 1B active parameters for low-resource chat and instruction following", family: "trinity", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-12-01", last_updated: "2026-05-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 }, license: "OpenMDW-1.1", links: [{ label: "Model card", url: "https://huggingface.co/arcee-ai/Trinity-Nano-Preview", type: "model_card" }, { label: "Announcement", url: "https://www.arcee.ai/blog/the-trinity-manifesto", type: "announcement" }, { label: "License", url: "https://huggingface.co/arcee-ai/Trinity-Nano-Preview/blob/main/LICENSE", type: "license" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/arcee-ai/Trinity-Nano-Preview", format: "safetensors" }] }, "google/gemini-3.1-flash-tts-preview": { id: "google/gemini-3.1-flash-tts-preview", name: "Gemini 3.1 Flash TTS Preview", description: "Low-latency speech generation with steerable prompts and expressive audio tags", family: "gemini-flash", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2026-04-15", last_updated: "2026-04-15", modalities: { input: ["text"], output: ["audio"] }, open_weights: false, limit: { context: 8192, output: 16384 } }, "google/gemma-4-26b-a4b-it": { id: "google/gemma-4-26b-a4b-it", name: "Gemma 4 26B A4B IT", description: "Open Gemma instruction model for efficient chat and self-hosted deployments", family: "gemma", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/google/gemma-4-26B-A4B-it" }] }, "google/gemini-3-pro-image": { id: "google/gemini-3-pro-image", name: "Nano Banana Pro", description: "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", family: "gemini-pro", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2026-05-28", last_updated: "2026-05-28", modalities: { input: ["text", "image"], output: ["text", "image"] }, open_weights: false, limit: { context: 65536, output: 32768 } }, "google/gemini-3-pro-image-preview": { id: "google/gemini-3-pro-image-preview", name: "Nano Banana Pro", description: "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", family: "gemini-pro", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2025-11-20", last_updated: "2025-11-20", modalities: { input: ["text", "image"], output: ["text", "image"] }, open_weights: false, limit: { context: 65536, output: 32768 } }, "google/gemini-flash-lite-latest": { id: "google/gemini-flash-lite-latest", name: "Gemini Flash-Lite Latest", description: "Fast Gemini model balancing multimodal reasoning, tool use, and cost", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-03", release_date: "2026-07-21", last_updated: "2026-07-21", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-2.0-flash": { id: "google/gemini-2.0-flash", name: "Gemini 2.0 Flash", description: "Earlier Gemini Flash workhorse for responsive multimodal apps and tool use", family: "gemini-flash", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-06", release_date: "2024-12-11", last_updated: "2024-12-11", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 8192 } }, "google/gemini-2.5-flash-tts": { id: "google/gemini-2.5-flash-tts", name: "Gemini 2.5 Flash TTS", description: "Speech generation model for controllable voice, narration, and audio delivery", family: "gemini-flash", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2025-09-30", last_updated: "2025-12-10", modalities: { input: ["text"], output: ["audio"] }, open_weights: false, limit: { context: 32768, output: 16384 } }, "google/gemini-3.5-flash-lite": { id: "google/gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite", description: "Fast Gemini model balancing multimodal reasoning, tool use, and cost", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-03", release_date: "2026-07-21", last_updated: "2026-07-21", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "SWE-Bench Pro", score: 54.2, metric: "resolve rate", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "Terminal-Bench", score: 54, metric: "accuracy", harness: "Terminus 2", version: "2.1", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "MLE-Bench", score: 39.2, metric: "average position score", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "GDPval-AA", score: 1140, metric: "Elo", version: "v2", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "OSWorld-Verified", score: 74, metric: "success rate", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "CharXiv Reasoning", score: 74.5, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "CharXiv Reasoning", score: 76.5, metric: "accuracy", variant: "with tools", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "GDM-MRCR", score: 72.2, metric: "accuracy", variant: "128k average, 8-needle", version: "v2", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "GDM-MRCR", score: 21.3, metric: "accuracy", variant: "1M pointwise, 8-needle", version: "v2", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }] }, "google/gemini-3.1-flash-lite-image": { id: "google/gemini-3.1-flash-lite-image", name: "Nano Banana 2 Lite", description: "Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: false, temperature: true, knowledge: "2025-01", release_date: "2026-06-30", last_updated: "2026-06-30", modalities: { input: ["text", "image"], output: ["text", "image"] }, open_weights: false, limit: { context: 65536, output: 4096 } }, "google/gemini-3.1-pro-preview": { id: "google/gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview", description: "Reasoning-first Gemini preview for agentic coding and complex problem solving", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-02-19", last_updated: "2026-02-19", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "SWE-Bench Pro", score: 54.2, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "Terminal-Bench", score: 70.3, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "SWE-Bench Pro", score: 46.1, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }, { name: "SWE-Atlas Codebase QnA", score: 13.5, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 33.81, metric: "score", harness: "Gemini CLI", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 29.84, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "Artificial Analysis Coding Agent Index", score: 43, metric: "average pass@1", harness: "Gemini CLI", variant: "high", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 45.6, metric: "pass@1", harness: "Gemini CLI", variant: "high", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 15.1, metric: "pass@1", harness: "Gemini CLI", variant: "high", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 68.3, metric: "pass@1", harness: "Gemini CLI", variant: "high", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "GPQA Diamond", score: 94.3, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 44.4, metric: "accuracy", dataset: "full set, text + MM", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "ARC-AGI-2", score: 77.1, metric: "accuracy", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "MMMU Pro", score: 80.5, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "MCP Atlas", score: 78.2, metric: "success rate", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "OSWorld-Verified", score: 76.2, metric: "success rate", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "CharXiv Reasoning", score: 83.3, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "GDPval-AA", score: 1314, metric: "Elo", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }] }, "google/gemini-3.1-pro-preview-customtools": { id: "google/gemini-3.1-pro-preview-customtools", name: "Gemini 3.1 Pro Preview Custom Tools", description: "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-02-19", last_updated: "2026-02-19", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/deep-research-preview-04-2026": { id: "google/deep-research-preview-04-2026", name: "Gemini Deep Research Preview", description: "Agentic model for autonomous multi-step research, synthesis, and cited reports", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2025-01", release_date: "2026-04-21", last_updated: "2026-04-21", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text", "image"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-2.5-flash-lite": { id: "google/gemini-2.5-flash-lite", name: "Gemini 2.5 Flash-Lite", description: "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2025-06-17", last_updated: "2025-06-17", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 9.5, metric: "index", source: "https://openrouter.ai/google/gemini-2.5-flash-lite/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 19.3, metric: "percent correct", source: "https://openrouter.ai/google/gemini-2.5-flash-lite/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 4.5, metric: "success rate", source: "https://openrouter.ai/google/gemini-2.5-flash-lite/benchmarks", date: "2026-03-11" }] }, "google/gemini-robotics-er-1.6-preview": { id: "google/gemini-robotics-er-1.6-preview", name: "Gemini Robotics-ER 1.6 Preview", description: "Vision-language model for embodied reasoning: spatial understanding, task planning, and physical-world agentic robotics", family: "gemini", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-04-14", last_updated: "2026-04-14", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 65536 } }, "google/gemma-4-31b-it": { id: "google/gemma-4-31b-it", name: "Gemma 4 31B IT", description: "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", family: "gemma", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/google/gemma-4-31B-it" }] }, "google/gemini-2.5-computer-use-preview-10-2025": { id: "google/gemini-2.5-computer-use-preview-10-2025", name: "Gemini 2.5 Computer Use Preview", description: "Specialized Gemini 2.5 model for browser-control agents that automate UI tasks", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-01", release_date: "2025-10-07", last_updated: "2025-10-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 64e3 } }, "google/lyria-3-clip-preview": { id: "google/lyria-3-clip-preview", name: "Lyria 3 Clip Preview", description: "Music generation model for short 30-second clips, loops, and previews from text or image prompts", family: "lyria", attachment: true, reasoning: false, tool_call: false, structured_output: false, temperature: true, release_date: "2026-03-25", last_updated: "2026-03-25", modalities: { input: ["text", "image"], output: ["text", "audio"] }, open_weights: false, limit: { context: 131072, output: 65536 } }, "google/veo-3.1-fast-generate-preview": { id: "google/veo-3.1-fast-generate-preview", name: "Veo 3.1 Fast Preview", description: "Video model for prompt-guided generation, editing, and motion workflows", family: "veo", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2025-10-15", last_updated: "2026-01-01", modalities: { input: ["text", "image", "video"], output: ["video"] }, open_weights: false, limit: { context: 1024, output: 0 } }, "google/gemini-embedding-001": { id: "google/gemini-embedding-001", name: "Gemini Embedding 001", description: "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", family: "gemini", attachment: false, reasoning: false, tool_call: false, temperature: false, knowledge: "2025-05", release_date: "2025-05-20", last_updated: "2025-05-20", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 2048, output: 1 } }, "google/gemini-flash-latest": { id: "google/gemini-flash-latest", name: "Gemini Flash Latest", description: "High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-03", release_date: "2026-08-13", last_updated: "2026-08-13", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-3.5-flash": { id: "google/gemini-3.5-flash", name: "Gemini 3.5 Flash", description: "Fast Gemini model balancing multimodal reasoning, tool use, and cost", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-05-19", last_updated: "2026-05-19", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "Terminal-Bench", score: 76.2, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "SWE-Bench Pro", score: 55.1, metric: "resolve rate", variant: "single attempt", dataset: "public", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "MCP Atlas", score: 83.6, metric: "success rate", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "Toolathlon", score: 56.5, metric: "success rate", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "OSWorld-Verified", score: 78.4, metric: "success rate", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "MMMU Pro", score: 83.6, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "CharXiv Reasoning", score: 84.2, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "Humanity's Last Exam", score: 40.2, metric: "accuracy", dataset: "full set, text + MM", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "ARC-AGI-2", score: 72.1, metric: "accuracy", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "GDPval-AA", score: 1656, metric: "Elo", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }] }, "google/gemini-3.5-live-translate-preview": { id: "google/gemini-3.5-live-translate-preview", name: "Gemini 3.5 Live Translate Preview", description: "Low-latency audio-to-audio model for real-time speech translation across 70+ languages", family: "gemini-pro", attachment: false, reasoning: false, tool_call: false, temperature: false, knowledge: "2025-01", release_date: "2026-06-09", last_updated: "2026-06-09", modalities: { input: ["audio"], output: ["audio", "text"] }, open_weights: false, limit: { context: 131072, output: 65536 } }, "google/veo-3.1-lite-generate-preview": { id: "google/veo-3.1-lite-generate-preview", name: "Veo 3.1 Lite Preview", description: "Video model for prompt-guided generation, editing, and motion workflows", family: "veo", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-03-31", last_updated: "2026-03-31", modalities: { input: ["text", "image"], output: ["video"] }, open_weights: false, limit: { context: 1024, output: 0 } }, "google/gemini-omni-flash-preview": { id: "google/gemini-omni-flash-preview", name: "Gemini Omni Flash Preview", description: "Video generation and editing model for fast, conversational text- and image-to-video workflows", family: "gemini", attachment: true, reasoning: true, tool_call: false, release_date: "2026-06-30", last_updated: "2026-06-30", modalities: { input: ["text", "image", "video"], output: ["video"] }, open_weights: false, limit: { context: 1048576, output: 57920 }, benchmarks: [{ name: "LMArena Text-to-Video Arena", score: 1527, metric: "Elo", source: "https://venturebeat.com/technology/googles-gemini-omni-flash-hits-the-api-turning-enterprise-video-production-into-a-conversation", date: "2026-06-30" }] }, "google/gemini-3.1-flash-lite": { id: "google/gemini-3.1-flash-lite", name: "Gemini 3.1 Flash Lite", description: "Low-latency Gemini model for high-volume multimodal and agent workloads", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-05-07", last_updated: "2026-05-07", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-2.5-flash": { id: "google/gemini-2.5-flash", name: "Gemini 2.5 Flash", description: "Fast Gemini workhorse for multimodal apps where latency and price matter", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2025-06-17", last_updated: "2025-06-17", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "Aider Polyglot", score: 55.1, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-25" }, { name: "Artificial Analysis Coding Index", score: 22.2, metric: "index", source: "https://openrouter.ai/google/gemini-2.5-flash/benchmarks", date: "2026-06-02" }, { name: "SciCode", score: 39.4, metric: "percent correct", source: "https://openrouter.ai/google/gemini-2.5-flash/benchmarks", date: "2026-06-02" }, { name: "Terminal-Bench Hard", score: 13.6, metric: "success rate", source: "https://openrouter.ai/google/gemini-2.5-flash/benchmarks", date: "2026-06-02" }] }, "google/gemini-3-pro-preview": { id: "google/gemini-3-pro-preview", name: "Gemini 3 Pro Preview", description: "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2025-11-18", last_updated: "2025-11-18", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "SWE-Bench Pro", score: 43.3, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "google/veo-3.1-generate-preview": { id: "google/veo-3.1-generate-preview", name: "Veo 3.1 Preview", description: "Video model for prompt-guided generation, editing, and motion workflows", family: "veo", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2025-10-15", last_updated: "2026-01", modalities: { input: ["text", "image"], output: ["video"] }, open_weights: false, limit: { context: 1024, output: 1 } }, "google/gemini-2.5-pro-tts": { id: "google/gemini-2.5-pro-tts", name: "Gemini 2.5 Pro TTS", description: "Speech generation model for controllable voice, narration, and audio delivery", family: "gemini-pro", attachment: false, reasoning: false, tool_call: false, temperature: false, knowledge: "2025-01", release_date: "2025-09-30", last_updated: "2025-12-10", modalities: { input: ["text"], output: ["audio"] }, open_weights: false, limit: { context: 32768, output: 16384 } }, "google/gemini-embedding-2": { id: "google/gemini-embedding-2", name: "Gemini Embedding 2", description: "Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space", family: "gemini", attachment: true, reasoning: false, tool_call: false, temperature: false, knowledge: "2025-11", release_date: "2026-04-22", last_updated: "2026-04-22", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 8192, output: 3072 } }, "google/gemma-4-E4B-it": { id: "google/gemma-4-E4B-it", name: "Gemma 4 E4B IT", description: "Open Gemma instruction model for efficient chat and self-hosted deployments", family: "gemma", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text", "image", "audio"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/google/gemma-4-E4B-it" }] }, "google/gemma-4-E2B-it": { id: "google/gemma-4-E2B-it", name: "Gemma 4 E2B IT", description: "Open Gemma instruction model for efficient chat and self-hosted deployments", family: "gemma", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text", "image", "audio"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/google/gemma-4-E2B-it" }] }, "google/gemini-3.6-flash": { id: "google/gemini-3.6-flash", name: "Gemini 3.6 Flash", description: "Fast Gemini model balancing multimodal reasoning, tool use, and cost", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-03", release_date: "2026-07-21", last_updated: "2026-07-21", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "SWE-Bench Pro", score: 58.7, metric: "resolve rate", harness: "Antigravity", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "DeepSWE", score: 49, metric: "resolve rate", variant: "high reasoning", version: "1.1", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "Terminal-Bench", score: 78, metric: "accuracy", harness: "Terminus 2", version: "2.1", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "MLE-Bench", score: 63.9, metric: "average position score", dataset: "Partial 30", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "GDPval-AA", score: 1421, metric: "Elo", version: "v2", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "OSWorld-Verified", score: 83, metric: "success rate", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "CharXiv Reasoning", score: 85.2, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "CharXiv Reasoning", score: 89.4, metric: "accuracy", variant: "with tools", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "GDM-MRCR", score: 91.8, metric: "accuracy", variant: "128k average, 8-needle", version: "v2", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "GDM-MRCR", score: 54, metric: "accuracy", variant: "1M pointwise, 8-needle", version: "v2", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }] }, "google/gemini-3.1-flash-image": { id: "google/gemini-3.1-flash-image", name: "Nano Banana 2", description: "Image model for prompt-driven generation, editing, and visual design workflows", family: "gemini-flash", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2026-05-28", last_updated: "2026-05-28", modalities: { input: ["text", "image", "video", "pdf"], output: ["text", "image"] }, open_weights: false, limit: { context: 131072, output: 32768 } }, "google/gemini-3.1-flash-image-preview": { id: "google/gemini-3.1-flash-image-preview", name: "Nano Banana 2", description: "Image model for prompt-driven generation, editing, and visual design workflows", family: "gemini-flash", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2026-02-26", last_updated: "2026-02-26", modalities: { input: ["text", "image", "pdf"], output: ["text", "image"] }, open_weights: false, limit: { context: 65536, output: 65536 } }, "google/gemini-2.0-flash-lite": { id: "google/gemini-2.0-flash-lite", name: "Gemini 2.0 Flash-Lite", description: "Low-latency Gemini model for high-volume multimodal and agent workloads", family: "gemini-flash-lite", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-06", release_date: "2024-12-11", last_updated: "2024-12-11", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 8192 } }, "google/lyria-3-pro-preview": { id: "google/lyria-3-pro-preview", name: "Lyria 3 Pro Preview", description: "Music generation model for full-length songs from text or images with vocals and structure", family: "lyria", attachment: true, reasoning: false, tool_call: false, structured_output: false, temperature: true, release_date: "2026-03-25", last_updated: "2026-03-25", modalities: { input: ["text", "image"], output: ["text", "audio"] }, open_weights: false, limit: { context: 131072, output: 8192 } }, "google/gemini-3.1-flash-lite-preview": { id: "google/gemini-3.1-flash-lite-preview", name: "Gemini 3.1 Flash Lite Preview", description: "Low-latency Gemini model for high-volume multimodal and agent workloads", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-03-03", last_updated: "2026-03-03", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-3-flash-preview": { id: "google/gemini-3-flash-preview", name: "Gemini 3 Flash Preview", description: "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2025-12-17", last_updated: "2025-12-17", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "SWE-Bench Pro", score: 34.63, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }, { name: "SWE-Atlas Codebase QnA", score: 8.2, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 10, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 30.3, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }] }, "google/gemini-2.5-flash-image": { id: "google/gemini-2.5-flash-image", name: "Nano Banana", description: "Nano Banana image model for fast generation, edits, and character-consistent assets", family: "gemini-flash", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2024-06", release_date: "2025-08-26", last_updated: "2025-08-26", modalities: { input: ["text", "image"], output: ["text", "image"] }, open_weights: false, limit: { context: 32768, output: 32768 } }, "google/gemini-2.5-pro": { id: "google/gemini-2.5-pro", name: "Gemini 2.5 Pro", description: "Google's proven reasoning model for coding, math, and multimodal analysis", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2025-06-17", last_updated: "2025-06-17", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "Aider Polyglot", score: 83.1, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-06-06" }, { name: "Artificial Analysis Coding Index", score: 32, metric: "index", source: "https://openrouter.ai/google/gemini-2.5-pro/benchmarks", date: "2026-06-02" }, { name: "SciCode", score: 42.8, metric: "percent correct", source: "https://openrouter.ai/google/gemini-2.5-pro/benchmarks", date: "2026-06-02" }, { name: "Terminal-Bench Hard", score: 26.5, metric: "success rate", source: "https://openrouter.ai/google/gemini-2.5-pro/benchmarks", date: "2026-06-02" }] }, "google/deep-research-max-preview-04-2026": { id: "google/deep-research-max-preview-04-2026", name: "Deep Research Max Preview", description: "Maximum-comprehensiveness agentic researcher for multi-step investigation, synthesis, and cited reports", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2025-01", release_date: "2026-04-21", last_updated: "2026-04-21", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text", "image"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-3.1-flash-live-preview": { id: "google/gemini-3.1-flash-live-preview", name: "Gemini 3.1 Flash Live Preview", description: "High-quality, low-latency Live API model for real-time dialogue and voice-first AI applications", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: false, temperature: true, knowledge: "2025-01", release_date: "2026-03-26", last_updated: "2026-03-26", modalities: { input: ["text", "image", "video", "audio"], output: ["text", "audio"] }, open_weights: false, limit: { context: 131072, output: 65536 } }, "google/gemini-3.7-flash": { id: "google/gemini-3.7-flash", name: "Gemini 3.7 Flash", description: "High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-03", release_date: "2026-08-13", last_updated: "2026-08-13", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "FrontierCode", score: 43.6, metric: "score", version: "1.1 Main", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }, { name: "DeepSWE", score: 65.3, metric: "resolve rate", version: "1.1", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }, { name: "Terminal-Bench", score: 85.8, metric: "accuracy", version: "2.1", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }, { name: "AutomationBench", score: 30.4, metric: "accuracy", dataset: "private set", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }, { name: "GDP.pdf", score: 34, metric: "accuracy", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }, { name: "GDM-MRCR", score: 97, metric: "accuracy", variant: "128k average, 8-needle", version: "v2", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }] }, "meta/llama-guard-3-8b": { id: "meta/llama-guard-3-8b", name: "Llama-Guard-3-8B", description: "Llama 3.1-based safety classifier for moderating prompts and model responses", family: "llama", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2023-12", release_date: "2024-07-23", last_updated: "2024-07-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-Guard-3-8B" }] }, "meta/llama-3.3-70b-instruct": { id: "meta/llama-3.3-70b-instruct", name: "Llama-3.3-70B-Instruct", description: "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", family: "llama", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2023-12", release_date: "2024-12-06", last_updated: "2024-12-06", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 10.7, metric: "index", source: "https://openrouter.ai/meta-llama/llama-3.3-70b-instruct/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 26, metric: "percent correct", source: "https://openrouter.ai/meta-llama/llama-3.3-70b-instruct/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 3, metric: "success rate", source: "https://openrouter.ai/meta-llama/llama-3.3-70b-instruct/benchmarks", date: "2026-03-11" }] }, "meta/llama-4-scout-17b-instruct": { id: "meta/llama-4-scout-17b-instruct", name: "Llama 4 Scout 17B Instruct", description: "Open Llama with long-context vision for efficient multimodal agents", family: "llama", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-08", release_date: "2025-04-05", last_updated: "2025-04-05", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 35e5, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E-Instruct" }] }, "meta/muse-glimmer-30b": { id: "meta/muse-glimmer-30b", name: "Muse Glimmer 30B", description: "Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.", family: "muse", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-01-04", release_date: "2026-08-10", last_updated: "2026-08-10", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 }, license: "Apache 2.0", links: [{ label: "Announcement", url: "https://research.meta.ai/blog/introducing-muse-glimmer-open-agentic-model", type: "announcement" }, { label: "Model card", url: "https://huggingface.co/meta-models/Muse-Glimmer-30B", type: "model_card" }, { label: "Developer docs", url: "https://developer.meta.com/ai/models/muse-glimmer/", type: "docs" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-models/Muse-Glimmer-30B" }], benchmarks: [{ name: "MCP Atlas", score: 75.5, metric: "success rate", variant: "public", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "DeepSearch QA", score: 74.6, source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "SWE-Bench Pro", score: 51.2, metric: "resolve rate", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "SWE-Bench Verified", score: 76, metric: "resolve rate", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "Terminal-Bench", score: 51.7, metric: "success rate", variant: "with terminus2", version: "2.1", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "OSWorld-Verified", score: 65.9, metric: "success rate", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "AIME 2026", score: 94.7, metric: "accuracy", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "GPQA Diamond", score: 83.5, metric: "accuracy", variant: "AA", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "CharXiv Reasoning", score: 78.8, metric: "accuracy", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }] }, "meta/llama-3.1-8b-instruct": { id: "meta/llama-3.1-8b-instruct", name: "Llama-3.1-8B-Instruct", description: "Compact open Llama model for lightweight chat, drafting, and self-hosting", family: "llama", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2023-12", release_date: "2024-07-23", last_updated: "2024-07-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct" }] }, "meta/muse-spark-1.2": { id: "meta/muse-spark-1.2", name: "Muse Spark 1.2", description: "Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.", family: "muse", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-05", last_updated: "2026-08-05", modalities: { input: ["text", "image", "video", "pdf", "audio"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 131072 } }, "meta/muse-spark-1.1": { id: "meta/muse-spark-1.1", name: "Muse Spark 1.1", description: "Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.", family: "muse", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-08", last_updated: "2026-07-09", modalities: { input: ["text", "image", "pdf", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 32e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 61.5, metric: "resolve rate", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "Terminal-Bench", score: 80, metric: "success rate", version: "2.1", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "DeepSWE", score: 53.3, metric: "resolve rate", version: "1.1", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "MCP Atlas", score: 88.1, metric: "success rate", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "JobBench", score: 54.7, metric: "success rate", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "Toolathlon-Verified", score: 75.6, metric: "success rate", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "Humanity's Last Exam", score: 62.1, metric: "accuracy", variant: "with tools", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "OSWorld-Verified", score: 80.8, metric: "success rate", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "Finance Agent", score: 57.2, metric: "accuracy", version: "v2", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "CharXiv Reasoning", score: 88.4, metric: "accuracy", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "BabyVision", score: 76.3, metric: "accuracy", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }] }, "meta/llama-3.2-1b": { id: "meta/llama-3.2-1b", name: "Llama-3.2-1B", description: "Compact open Llama base model for lightweight and on-device use", family: "llama", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2023-12", release_date: "2024-09-25", last_updated: "2024-09-25", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "Llama 3.2 Community License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-3.2-1B" }] }, "meta/llama-4-maverick-17b-instruct": { id: "meta/llama-4-maverick-17b-instruct", name: "Llama 4 Maverick 17B Instruct", description: "Open multimodal Llama for strong reasoning with efficient everyday serving", family: "llama", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-08", release_date: "2025-04-05", last_updated: "2025-04-05", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct" }], benchmarks: [{ name: "Aider Polyglot", score: 15.6, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-04-06" }, { name: "SWE-Bench Pro", score: 5.24, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "meta/llama-3.2-3b": { id: "meta/llama-3.2-3b", name: "Llama-3.2-3B", description: "Small open Llama base model for lightweight text generation and self-hosting", family: "llama", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2023-12", release_date: "2024-09-25", last_updated: "2024-09-25", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "Llama 3.2 Community License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-3.2-3B" }] }, "meta/llama-3.2-11b-vision-instruct": { id: "meta/llama-3.2-11b-vision-instruct", name: "Llama-3.2-11B-Vision-Instruct", description: "Open multimodal Llama model for image understanding, captioning, and visual QA", family: "llama", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2023-12", release_date: "2024-09-25", last_updated: "2024-09-25", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct" }] }, "sdaia/allam-2-7b": { id: "sdaia/allam-2-7b", name: "ALLaM-2-7b", description: "ALLaM-2-7b instruction tuned model by SDAIA", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2025-01-23", last_updated: "2025-01-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 4096, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/ALLaM-AI/ALLaM-2.0-7B-Instruct" }] }, "poolside/laguna-m.1": { id: "poolside/laguna-m.1", name: "Laguna M.1", description: "Poolside's open-weight model for agentic coding and long-horizon work", family: "laguna", attachment: false, reasoning: true, tool_call: true, structured_output: false, temperature: true, release_date: "2026-04-28", last_updated: "2026-06-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 } }, "poolside/laguna-s-2.1": { id: "poolside/laguna-s-2.1", name: "Laguna S 2.1", description: "Agentic coding model from Poolside in the XS size class for local deployment", family: "laguna", attachment: false, reasoning: true, tool_call: true, structured_output: false, temperature: true, release_date: "2026-07-21", last_updated: "2026-07-21", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 32768 } }, "poolside/laguna-xs-2.1": { id: "poolside/laguna-xs-2.1", name: "Laguna XS 2.1", description: "Agentic coding model from Poolside in the XS size class for local deployment", family: "laguna", attachment: false, reasoning: true, tool_call: true, structured_output: false, temperature: true, release_date: "2026-07-02", last_updated: "2026-07-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 }, benchmarks: [{ name: "SWE-Bench Verified", score: 70.9, metric: "resolved", harness: "Harbor", source: "https://poolside.ai/blog/introducing-laguna-xs-2-1", date: "2026-07-02" }, { name: "SWE-Bench Multilingual", score: 63.1, metric: "resolve rate", harness: "Harbor", source: "https://poolside.ai/blog/introducing-laguna-xs-2-1", date: "2026-07-02" }, { name: "SWE-Bench Pro", score: 47.6, metric: "resolve rate", harness: "Harbor", source: "https://poolside.ai/blog/introducing-laguna-xs-2-1", date: "2026-07-02" }, { name: "Terminal-Bench", score: 37.5, metric: "success rate", harness: "Harbor", version: "2.0", source: "https://poolside.ai/blog/introducing-laguna-xs-2-1", date: "2026-07-02" }] }, "poolside/laguna-xs.2": { id: "poolside/laguna-xs.2", name: "Laguna XS.2", description: "Agentic coding model from Poolside in the XS size class for local deployment", family: "laguna", attachment: false, reasoning: true, tool_call: true, structured_output: false, temperature: true, release_date: "2026-04-28", last_updated: "2026-06-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 } }, "bytedance-seed/seed-2.0-code": { id: "bytedance-seed/seed-2.0-code", name: "Seed 2.0 Code", description: "ByteDance Seed coding model for multimodal software engineering and long-running agents", family: "seed", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-14", last_updated: "2026-02-14", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 131072 } }, "bytedance-seed/seed-2.1-turbo": { id: "bytedance-seed/seed-2.1-turbo", name: "Seed 2.1 Turbo", description: "Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-23", last_updated: "2026-06-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 256e3 } }, "bytedance-seed/seed-1-6": { id: "bytedance-seed/seed-1-6", name: "Seed 1.6", description: "ByteDance Seed model for long-context reasoning, instruction following, and tool-assisted tasks", family: "seed", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-10-15", last_updated: "2025-10-15", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 64e3 } }, "bytedance-seed/seed-character": { id: "bytedance-seed/seed-character", name: "Seed Character", description: "ByteDance Seed model optimized for character-driven dialogue and consistent conversational behavior", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-23", last_updated: "2026-06-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 256e3 } }, "bytedance-seed/seed-1-6-vision": { id: "bytedance-seed/seed-1-6-vision", name: "Seed 1.6 Vision", description: "ByteDance Seed multimodal model for image understanding, visual reasoning, and tool-assisted tasks", family: "seed", attachment: true, reasoning: false, tool_call: true, temperature: true, release_date: "2025-08-15", last_updated: "2025-08-15", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 32e3 } }, "bytedance-seed/seed-2.0-pro": { id: "bytedance-seed/seed-2.0-pro", name: "Seed 2.0 Pro", description: "Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-14", last_updated: "2026-02-14", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 128e3 } }, "bytedance-seed/seed-evolving": { id: "bytedance-seed/seed-evolving", name: "Seed Evolving", description: "Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-23", last_updated: "2026-06-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 256e3 } }, "bytedance-seed/seed-2.1-pro": { id: "bytedance-seed/seed-2.1-pro", name: "Seed 2.1 Pro", description: "Flagship ByteDance Seed 2.1 model for complex multimodal reasoning, coding, and agents", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-23", last_updated: "2026-06-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 256e3 } }, "bytedance-seed/seed-1-6-flash": { id: "bytedance-seed/seed-1-6-flash", name: "Seed 1.6 Flash", description: "Low-latency ByteDance Seed model for high-throughput chat, extraction, and lightweight tool use", family: "seed", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-08-28", last_updated: "2025-08-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 32e3 } }, "bytedance-seed/seed-2.0-mini": { id: "bytedance-seed/seed-2.0-mini", name: "Seed 2.0 Mini", description: "Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-14", last_updated: "2026-02-14", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 32e3 } }, "bytedance-seed/seed-1-8": { id: "bytedance-seed/seed-1-8", name: "Seed 1.8", description: "ByteDance Seed model for multimodal reasoning, long-context analysis, and agent workflows", family: "seed", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-12-28", last_updated: "2025-12-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 64e3 } }, "bytedance-seed/seed-2.0-lite": { id: "bytedance-seed/seed-2.0-lite", name: "Seed 2.0 Lite", description: "Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-14", last_updated: "2026-02-14", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 32e3 } }, "moonshotai/kimi-k2.7-code": { id: "moonshotai/kimi-k2.7-code", name: "Kimi K2.7 Code", description: "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", family: "kimi-k2", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-01", release_date: "2026-06-12", last_updated: "2026-06-12", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2.7-Code" }], benchmarks: [{ name: "Kimi Code Bench", score: 62, harness: "Kimi Code CLI", version: "v2", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }, { name: "Program Bench", score: 53.6, harness: "Kimi Code CLI", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }, { name: "MLS Bench Lite", score: 35.1, harness: "Kimi Code CLI", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }, { name: "MCP Atlas", score: 76, metric: "success rate", harness: "Kimi Code CLI", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }, { name: "MCP Mark Verified", score: 81.1, metric: "success rate", harness: "Kimi Code CLI", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }, { name: "Kimi Claw 24/7 Bench", score: 46.9, harness: "Kimi Code CLI", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }] }, "moonshotai/kimi-k3": { id: "moonshotai/kimi-k3", name: "Kimi K3", description: "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work", family: "kimi-k3", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, release_date: "2026-07-16", last_updated: "2026-07-16", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 131072 }, benchmarks: [{ name: "DeepSWE", score: 67.5, metric: "resolve rate", harness: "Kimi Code", variant: "max effort", version: "1.1", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "Terminal-Bench", score: 88.3, metric: "accuracy", harness: "Kimi Code", variant: "max effort", version: "2.1", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "FrontierSWE", score: 81.2, metric: "dominance score", harness: "Kimi Code", variant: "max effort", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "Program Bench", score: 77.8, metric: "score", harness: "Kimi Code", variant: "max effort", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "SWE Marathon", score: 42, metric: "resolve rate", harness: "Claude Code", variant: "max effort", version: "1.1", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "GDPval-AA", score: 1668, metric: "Elo", variant: "max effort", version: "v2", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "AA-Briefcase", score: 1548, metric: "Elo", variant: "max effort", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "AutomationBench", score: 30.8, metric: "success rate", variant: "max effort", dataset: "600-task public subset", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "JobBench", score: 52.9, metric: "score", variant: "max effort", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "SpreadsheetBench", score: 34.8, metric: "score", harness: "Claude Code", variant: "max effort", version: "2", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "BrowseComp", score: 91.2, metric: "accuracy", variant: "max effort, context compaction", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "CharXiv Reasoning", score: 91.3, metric: "accuracy", variant: "max effort, with tools", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "ZeroBench", score: 41, metric: "pass@5", variant: "max effort, with tools", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }] }, "moonshotai/kimi-k2-thinking-turbo": { id: "moonshotai/kimi-k2-thinking-turbo", name: "Kimi K2 Thinking Turbo", description: "Kimi reasoning model for long-horizon research, planning, and tool use", family: "kimi-thinking", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-08", release_date: "2025-11-06", last_updated: "2025-11-06", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2-Thinking" }] }, "moonshotai/kimi-k2.5": { id: "moonshotai/kimi-k2.5", name: "Kimi K2.5", description: "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", family: "kimi-k2", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-01", release_date: "2026-01", last_updated: "2026-01", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2.5" }], benchmarks: [{ name: "SWE-Bench Verified", score: 70.8, metric: "resolved", source: "https://www.swebench.com/" }, { name: "SWE-Atlas Codebase QnA", score: 13.1, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 20.95, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 25.77, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }] }, "moonshotai/kimi-k2.7-code-highspeed": { id: "moonshotai/kimi-k2.7-code-highspeed", name: "Kimi K2.7 Code Highspeed", description: "Lower-latency Kimi Code variant for interactive edits and coding-agent loops", family: "kimi-k2", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-01", release_date: "2026-06-12", last_updated: "2026-06-12", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2.7-Code" }] }, "moonshotai/kimi-k2-thinking": { id: "moonshotai/kimi-k2-thinking", name: "Kimi K2 Thinking", description: "Thinking Kimi model for slower research passes, planning, and hard technical questions", family: "kimi-thinking", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-08", release_date: "2025-11-06", last_updated: "2025-11-06", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2-Thinking" }], benchmarks: [{ name: "SWE-Bench Verified", score: 71.3, metric: "resolved", source: "https://huggingface.co/moonshotai/Kimi-K2-Thinking" }] }, "moonshotai/kimi-k2.6": { id: "moonshotai/kimi-k2.6", name: "Kimi K2.6", description: "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", family: "kimi-k2", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-04-21", last_updated: "2026-04-21", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2.6" }], benchmarks: [{ name: "SWE-Bench Verified", score: 80.2, metric: "resolved", source: "https://huggingface.co/moonshotai/Kimi-K2.6" }, { name: "Artificial Analysis Coding Agent Index", score: 50.5, metric: "average pass@1", harness: "Claude Code", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 59.8, metric: "pass@1", harness: "Claude Code", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 27.3, metric: "pass@1", harness: "Claude Code", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 64.3, metric: "pass@1", harness: "Claude Code", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }] }, "anthropic/claude-opus-4-20250514": { id: "anthropic/claude-opus-4-20250514", name: "Claude Opus 4", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-05-22", last_updated: "2025-05-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 32e3 }, benchmarks: [{ name: "Aider Polyglot", score: 72, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-25" }] }, "anthropic/claude-opus-4-7": { id: "anthropic/claude-opus-4-7", name: "Claude Opus 4.7", description: "Stronger Opus tier for advanced software work and high-stakes reasoning", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2026-01-31", release_date: "2026-04-16", last_updated: "2026-04-16", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 64.3, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "Terminal-Bench", score: 66.1, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "SWE-Atlas Refactoring", score: 48.57, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "Artificial Analysis Coding Agent Index", score: 66.6, metric: "average pass@1", harness: "Claude Code", variant: "max", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 81, metric: "pass@1", harness: "Claude Code", variant: "max", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 44.9, metric: "pass@1", harness: "Claude Code", variant: "max", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 73.8, metric: "pass@1", harness: "Claude Code", variant: "max", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Artificial Analysis Coding Agent Index", score: 61.2, metric: "average pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 78.4, metric: "pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 34.4, metric: "pass@1", harness: "Cursor CLI", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 70.6, metric: "pass@1", harness: "Cursor CLI", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Artificial Analysis Coding Agent Index", score: 59.9, metric: "average pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 71.7, metric: "pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 36.4, metric: "pass@1", harness: "Claude Code", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 71.4, metric: "pass@1", harness: "Claude Code", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "GPQA Diamond", score: 94.2, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 46.9, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 54.7, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "OSWorld-Verified", score: 78, metric: "success rate", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }] }, "anthropic/claude-mythos-5": { id: "anthropic/claude-mythos-5", name: "Claude Mythos 5", description: "Restricted Claude model for advanced cybersecurity and biology research workflows", family: "claude-mythos", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2026-01-31", release_date: "2026-06-09", last_updated: "2026-06-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 } }, "anthropic/claude-opus-4-1-20250805": { id: "anthropic/claude-opus-4-1-20250805", name: "Claude Opus 4.1", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-08-05", last_updated: "2025-08-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 32e3 } }, "anthropic/claude-3-5-sonnet-20241022": { id: "anthropic/claude-3-5-sonnet-20241022", name: "Claude Sonnet 3.5 v2", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04-30", release_date: "2024-10-22", last_updated: "2024-10-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 8192 }, benchmarks: [{ name: "Aider Polyglot", score: 51.6, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-01-17" }] }, "anthropic/claude-opus-4-8": { id: "anthropic/claude-opus-4-8", name: "Claude Opus 4.8", description: "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2026-01", release_date: "2026-05-28", last_updated: "2026-05-28", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 69.2, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "Terminal-Bench", score: 74.6, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "SWE-Bench Verified", score: 88.6, metric: "resolved", source: "https://benchlm.ai/benchmarks/sweVerified" }, { name: "Humanity's Last Exam", score: 49.8, metric: "accuracy", variant: "no tools", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "Humanity's Last Exam", score: 57.9, metric: "accuracy", variant: "with tools", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "OSWorld-Verified", score: 83.4, metric: "success rate", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "FrontierCode", score: 13.4, metric: "pass rate", variant: "high effort", dataset: "Diamond", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }] }, "anthropic/claude-3-5-haiku-20241022": { id: "anthropic/claude-3-5-haiku-20241022", name: "Claude Haiku 3.5", description: "Fast Claude model for responsive assistance, classification, and lightweight agents", family: "claude-haiku", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-07-31", release_date: "2024-10-22", last_updated: "2024-10-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 8192 }, benchmarks: [{ name: "Aider Polyglot", score: 28, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-21" }] }, "anthropic/claude-opus-4-1": { id: "anthropic/claude-opus-4-1", name: "Claude Opus 4.1 (latest)", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-08-05", last_updated: "2025-08-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 32e3 } }, "anthropic/claude-sonnet-5": { id: "anthropic/claude-sonnet-5", name: "Claude Sonnet 5", description: "Everyday Claude agent model for coding, planning, browsing, and general work", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2026-01-31", release_date: "2026-06-30", last_updated: "2026-06-30", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Verified", score: 85.2, metric: "resolved", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "SWE-Bench Pro", score: 63.2, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "SWE-Bench Multilingual", score: 78.3, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "Terminal-Bench", score: 80.4, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "OSWorld-Verified", score: 81.2, metric: "success rate", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "BrowseComp", score: 84.7, metric: "accuracy", variant: "single agent", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "FrontierCode", score: 38.8, metric: "pass rate", version: "v1", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }] }, "anthropic/claude-3-7-sonnet-20250219": { id: "anthropic/claude-3-7-sonnet-20250219", name: "Claude Sonnet 3.7", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-10-31", release_date: "2025-02-19", last_updated: "2025-02-19", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "Aider Polyglot", score: 64.9, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-02-24" }] }, "anthropic/claude-sonnet-4-5-20250929": { id: "anthropic/claude-sonnet-4-5-20250929", name: "Claude Sonnet 4.5", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-07-31", release_date: "2025-09-29", last_updated: "2025-09-29", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 } }, "anthropic/claude-opus-4-5": { id: "anthropic/claude-opus-4-5", name: "Claude Opus 4.5 (latest)", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-05", release_date: "2025-11-24", last_updated: "2025-11-24", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 } }, "anthropic/claude-sonnet-4-6": { id: "anthropic/claude-sonnet-4-6", name: "Claude Sonnet 4.6", description: "Claude workhorse for coding agents, careful analysis, and production cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-08-31", release_date: "2026-02-17", last_updated: "2026-03-13", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 64e3 }, benchmarks: [{ name: "SWE-Atlas Codebase QnA", score: 31.2, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 32.21, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 31.76, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "Artificial Analysis Coding Agent Index", score: 49.4, metric: "average pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 70.3, metric: "pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 14.9, metric: "pass@1", harness: "Claude Code", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 63.1, metric: "pass@1", harness: "Claude Code", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 67, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "Humanity's Last Exam", score: 34.6, metric: "accuracy", variant: "no tools", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "Humanity's Last Exam", score: 46.8, metric: "accuracy", variant: "with tools", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "OSWorld-Verified", score: 78.5, metric: "success rate", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }] }, "anthropic/claude-opus-4-0": { id: "anthropic/claude-opus-4-0", name: "Claude Opus 4 (latest)", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-05-22", last_updated: "2025-05-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 32e3 }, benchmarks: [{ name: "Aider Polyglot", score: 72, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-25" }] }, "anthropic/claude-haiku-4-5-20251001": { id: "anthropic/claude-haiku-4-5-20251001", name: "Claude Haiku 4.5", description: "Fast Claude model for responsive assistance, classification, and lightweight agents", family: "claude-haiku", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-02-28", release_date: "2025-10-15", last_updated: "2025-10-15", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 } }, "anthropic/claude-sonnet-4-5": { id: "anthropic/claude-sonnet-4-5", name: "Claude Sonnet 4.5 (latest)", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-07-31", release_date: "2025-09-29", last_updated: "2025-09-29", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 43.6, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "anthropic/claude-opus-4-5-20251101": { id: "anthropic/claude-opus-4-5-20251101", name: "Claude Opus 4.5", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-05", release_date: "2025-11-01", last_updated: "2025-11-01", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 45.89, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "anthropic/claude-fable-5": { id: "anthropic/claude-fable-5", name: "Claude Fable 5", description: "Claude model for creative writing, analysis, and controlled agent workflows", family: "claude-fable", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2026-01-31", release_date: "2026-06-09", last_updated: "2026-06-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 80.3, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "SWE-Bench Verified", score: 95, metric: "resolved", source: "https://benchlm.ai/benchmarks/sweVerified" }, { name: "Terminal-Bench", score: 88, metric: "success rate", version: "2.1", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "Humanity's Last Exam", score: 59, metric: "accuracy", variant: "no tools", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "Humanity's Last Exam", score: 64.5, metric: "accuracy", variant: "with tools", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "OSWorld-Verified", score: 85, metric: "success rate", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "FrontierCode", score: 29.3, metric: "pass rate", variant: "high effort", dataset: "Diamond", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "GDPval-AA", score: 1932, metric: "Elo", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "AutomationBench", score: 17.4, metric: "success rate", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }] }, "anthropic/claude-sonnet-4-20250514": { id: "anthropic/claude-sonnet-4-20250514", name: "Claude Sonnet 4", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-05-22", last_updated: "2025-05-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "Aider Polyglot", score: 61.3, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-24" }] }, "anthropic/claude-sonnet-4-0": { id: "anthropic/claude-sonnet-4-0", name: "Claude Sonnet 4 (latest)", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-05-22", last_updated: "2025-05-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "Aider Polyglot", score: 61.3, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-24" }, { name: "SWE-Bench Pro", score: 42.7, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "anthropic/claude-haiku-4-5": { id: "anthropic/claude-haiku-4-5", name: "Claude Haiku 4.5 (latest)", description: "Fast Claude lane for lightweight agents, office tasks, and responsive chat", family: "claude-haiku", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-02-28", release_date: "2025-10-15", last_updated: "2025-10-15", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 39.45, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "anthropic/claude-opus-4-6": { id: "anthropic/claude-opus-4-6", name: "Claude Opus 4.6", description: "High-end Claude for difficult coding, planning, and slower expert reasoning", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-05-31", release_date: "2026-02-05", last_updated: "2026-03-13", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 51.9, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }, { name: "SWE-Atlas Codebase QnA", score: 33.3, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Codebase QnA", score: 30, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 35.58, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 36.67, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "SWE-Atlas Test Writing", score: 36.08, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "Artificial Analysis Coding Agent Index", score: 51.3, metric: "average pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 71.9, metric: "pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 11.8, metric: "pass@1", harness: "Claude Code", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 70.2, metric: "pass@1", harness: "Claude Code", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }] }, "anthropic/claude-3-haiku-20240307": { id: "anthropic/claude-3-haiku-20240307", name: "Claude Haiku 3", description: "Legacy model retained for compatibility with older integrations", family: "claude-haiku", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2023-08-31", release_date: "2024-03-13", last_updated: "2024-03-13", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 4096 } }, "anthropic/claude-opus-5": { id: "anthropic/claude-opus-5", name: "Claude Opus 5", description: "Strongest Claude Opus model for coding, agents, and professional work", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2026-05", release_date: "2026-07-24", last_updated: "2026-07-24", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Verified", score: 96, metric: "resolved", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "SWE-Bench Pro", score: 79.2, metric: "resolve rate", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "SWE-Bench Multilingual", score: 89.5, metric: "resolve rate", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "SWE-Bench Multimodal", score: 59.4, metric: "resolve rate", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "DeepSWE", score: 68.8, metric: "resolve rate", version: "1.1", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "FrontierCode", score: 53.4, metric: "mean@5", variant: "medium effort", dataset: "Main", version: "1.1", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "Frontier-Bench", score: 43.3, metric: "mean reward", harness: "mini-SWE-agent", variant: "max effort", version: "v0.1", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "BrowseComp", score: 90.8, metric: "accuracy", variant: "single agent", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "Humanity's Last Exam", score: 56.3, metric: "accuracy", variant: "no tools", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "Humanity's Last Exam", score: 64.7, metric: "accuracy", variant: "with tools", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "DeepSearchQA", score: 95, metric: "F1", variant: "max effort", source: "https://www.anthropic.com/news/claude-opus-5", date: "2026-07-24" }, { name: "OSWorld", score: 70.6, metric: "success rate", version: "2.0", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "GDPval-AA", score: 1861, metric: "Elo", variant: "max effort", version: "v2", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "AA-Briefcase", score: 1720, metric: "Elo", variant: "max effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "AutomationBench", score: 26, metric: "success rate", variant: "max effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "ARC-AGI-1", score: 97.5, metric: "accuracy", variant: "max effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "ARC-AGI-2", score: 90.4, metric: "accuracy", variant: "max effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "ARC-AGI-3", score: 30.2, metric: "RHAE", variant: "high effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "HealthBench Professional", score: 59.8, metric: "score", variant: "max effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }] }, "cohere/c4ai-aya-expanse-32b": { id: "cohere/c4ai-aya-expanse-32b", name: "Aya Expanse 32B", description: "Open multilingual model optimized for generation across 23 languages", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2024-10-24", last_updated: "2024-10-24", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4e3 }, license: "CC-BY-NC-4.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/aya-expanse-32b" }] }, "cohere/command-a-reasoning-08-2025": { id: "cohere/command-a-reasoning-08-2025", name: "Command A Reasoning", description: "Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows", family: "command-a", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2025-08-21", last_updated: "2025-08-21", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 32e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-a-reasoning-08-2025" }] }, "cohere/c4ai-aya-vision-32b": { id: "cohere/c4ai-aya-vision-32b", name: "Aya Vision 32B", description: "Open multilingual vision model for OCR, visual reasoning, and image question answering", attachment: true, reasoning: false, tool_call: false, temperature: true, release_date: "2025-03-04", last_updated: "2025-05-14", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 16e3, output: 4e3 }, license: "CC-BY-NC-4.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/aya-vision-32b" }] }, "cohere/command-r-plus-08-2024": { id: "cohere/command-r-plus-08-2024", name: "Command R+", description: "Cohere's RAG workhorse for long-context enterprise search and tool use", family: "command-r", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2024-08-30", last_updated: "2024-08-30", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-r-plus-08-2024" }] }, "cohere/command-a-translate-08-2025": { id: "cohere/command-a-translate-08-2025", name: "Command A Translate", description: "Translation model for multilingual conversion, localization, and cross-language workflows", family: "command-a", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2025-08-28", last_updated: "2025-08-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 8e3, output: 8e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-a-translate-08-2025" }] }, "cohere/command-r7b-arabic-02-2025": { id: "cohere/command-r7b-arabic-02-2025", name: "Command R7B Arabic", description: "Open Command R model optimized for Arabic enterprise chat, RAG, and cultural knowledge", family: "command-r", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2025-02-27", last_updated: "2025-02-27", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-r7b-arabic-02-2025" }] }, "cohere/c4ai-aya-expanse-8b": { id: "cohere/c4ai-aya-expanse-8b", name: "Aya Expanse 8B", description: "Compact open multilingual model optimized for generation across 23 languages", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2024-10-24", last_updated: "2024-10-24", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 8e3, output: 4e3 }, license: "CC-BY-NC-4.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/aya-expanse-8b" }] }, "cohere/command-a-vision-07-2025": { id: "cohere/command-a-vision-07-2025", name: "Command A Vision", description: "Cohere vision model for multilingual document analysis, OCR, and image understanding", family: "command-a", attachment: true, reasoning: false, tool_call: false, temperature: true, knowledge: "2024-06-01", release_date: "2025-07-31", last_updated: "2025-07-31", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-a-vision-07-2025" }] }, "cohere/command-a-plus-05-2026": { id: "cohere/command-a-plus-05-2026", name: "Command A Plus", description: "Cohere's stronger command model for multilingual agents and enterprise workflows", family: "command-a", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-04-01", release_date: "2026-05-20", last_updated: "2026-06-09", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 64e3 } }, "cohere/command-a-03-2025": { id: "cohere/command-a-03-2025", name: "Command A", description: "Cohere command model for multilingual enterprise agents, tools, and chat", family: "command-a", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2025-03-13", last_updated: "2025-03-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 8e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-a-03-2025" }], benchmarks: [{ name: "Aider Polyglot", score: 12, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-03-14" }] }, "cohere/command-r7b-12-2024": { id: "cohere/command-r7b-12-2024", name: "Command R7B", description: "Cohere retrieval model for long-context chat and enterprise RAG workflows", family: "command-r", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2024-12-02", last_updated: "2024-12-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-r7b-12-2024" }] }, "cohere/command-r-08-2024": { id: "cohere/command-r-08-2024", name: "Command R", description: "Cohere retrieval model for long-context chat and enterprise RAG workflows", family: "command-r", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2024-08-30", last_updated: "2024-08-30", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-r-08-2024" }] }, "cohere/north-mini-code-1-0": { id: "cohere/north-mini-code-1-0", name: "North Mini Code", description: "Cohere coding model for practical software engineering and agentic edits", family: "north", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-09-23", release_date: "2026-06-09", last_updated: "2026-06-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 64e3 }, benchmarks: [{ name: "SWE-Bench Verified", score: 67.6, metric: "resolved", harness: "SWE-agent", source: "https://huggingface.co/CohereLabs/North-Mini-Code-1.0", date: "2026-06-09" }, { name: "SWE-Bench Pro", score: 40.2, metric: "resolve rate", harness: "SWE-agent", source: "https://huggingface.co/CohereLabs/North-Mini-Code-1.0", date: "2026-06-09" }, { name: "Artificial Analysis Intelligence Index", score: 27.6, metric: "index score", source: "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model", date: "2026-06-09" }, { name: "Artificial Analysis Coding Index", score: 33.4, metric: "index score", source: "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model", date: "2026-06-09" }, { name: "GDPval-AA", score: 14, metric: "win rate", source: "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model", date: "2026-06-09" }, { name: "\u03C4\xB2-Bench Telecom", score: 37, metric: "success rate", source: "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model", date: "2026-06-09" }] }, "cohere/c4ai-aya-vision-8b": { id: "cohere/c4ai-aya-vision-8b", name: "Aya Vision 8B", description: "Compact open multilingual vision model for OCR and visual question answering", attachment: true, reasoning: false, tool_call: false, temperature: true, release_date: "2025-03-04", last_updated: "2025-05-14", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 16e3, output: 4e3 }, license: "CC-BY-NC-4.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/aya-vision-8b" }] }, "openai/gpt-4o": { id: "openai/gpt-4o", name: "GPT-4o", description: "Omni-era GPT for multimodal chat, practical coding, and general assistants", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2023-09", release_date: "2024-05-13", last_updated: "2024-08-06", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 }, benchmarks: [{ name: "Aider Polyglot", score: 23.1, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-30" }] }, "openai/gpt-image-1.5": { id: "openai/gpt-image-1.5", name: "GPT-Image-1.5", description: "Image model for prompt-driven generation, editing, and visual design workflows", family: "gpt-image", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2025-11-25", last_updated: "2025-11-25", modalities: { input: ["text", "image"], output: ["text", "image"] }, open_weights: false, limit: { context: 0, output: 0 } }, "openai/gpt-5.3-chat-latest": { id: "openai/gpt-5.3-chat-latest", name: "GPT-5.3 Chat (latest)", description: "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-08-31", release_date: "2026-03-03", last_updated: "2026-03-03", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 } }, "openai/gpt-5-nano": { id: "openai/gpt-5-nano", name: "GPT-5 Nano", description: "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", family: "gpt-nano", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05-30", release_date: "2025-08-07", last_updated: "2025-08-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-4o-mini": { id: "openai/gpt-4o-mini", name: "GPT-4o mini", description: "Small omni GPT for cheap multimodal assistance and production-scale traffic", family: "gpt-mini", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2023-09", release_date: "2024-07-18", last_updated: "2024-07-18", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 }, benchmarks: [{ name: "Aider Polyglot", score: 3.6, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-21" }, { name: "SciCode", score: 22.9, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-4o-mini/benchmarks", date: "2026-03-11" }] }, "openai/gpt-3.5-turbo": { id: "openai/gpt-3.5-turbo", name: "GPT-3.5-turbo", description: "Compact GPT model for low-latency assistance and high-volume workloads", family: "gpt", attachment: false, reasoning: false, tool_call: false, structured_output: false, temperature: true, knowledge: "2021-09-01", release_date: "2023-03-01", last_updated: "2023-11-06", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 16385, output: 4096 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 10.7, metric: "index", source: "https://openrouter.ai/openai/gpt-3.5-turbo/benchmarks", date: "2026-03-11" }] }, "openai/o1-pro": { id: "openai/o1-pro", name: "o1-pro", description: "O-series reasoning model for hard analysis, math, coding, and planning", family: "o-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2023-09", release_date: "2025-03-19", last_updated: "2025-03-19", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 } }, "openai/whisper-large-v3": { id: "openai/whisper-large-v3", name: "Whisper 3 Large", description: "Open Whisper checkpoint for robust multilingual transcription and captioning", family: "whisper", attachment: false, reasoning: false, tool_call: false, release_date: "2024-10-01", last_updated: "2024-10-01", modalities: { input: ["audio"], output: ["text"] }, open_weights: true, limit: { context: 448, output: 4096 } }, "openai/gpt-5.5-pro": { id: "openai/gpt-5.5-pro", name: "GPT-5.5 Pro", description: "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", family: "gpt-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-12-01", release_date: "2026-04-23", last_updated: "2026-04-23", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "BrowseComp", score: 90.1, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 43.1, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 57.2, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 52.4, metric: "accuracy", dataset: "Tier 1-3", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 39.6, metric: "accuracy", dataset: "Tier 4", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GDPval", score: 82.3, metric: "wins or ties", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GeneBench", score: 33.2, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }] }, "openai/gpt-4-turbo": { id: "openai/gpt-4-turbo", name: "GPT-4 Turbo", description: "Compact GPT model for low-latency assistance and high-volume workloads", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: false, temperature: true, knowledge: "2023-12", release_date: "2023-11-06", last_updated: "2024-04-09", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 4096 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 21.5, metric: "index", source: "https://openrouter.ai/openai/gpt-4-turbo/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 31.9, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-4-turbo/benchmarks", date: "2026-03-11" }] }, "openai/gpt-4": { id: "openai/gpt-4", name: "GPT-4", description: "GPT model for general reasoning, writing, coding, and tool-assisted tasks", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: false, temperature: true, knowledge: "2023-11", release_date: "2023-11-06", last_updated: "2024-04-09", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 8192, output: 8192 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 13.1, metric: "index", source: "https://openrouter.ai/openai/gpt-4/benchmarks", date: "2026-03-11" }] }, "openai/gpt-5-chat-latest": { id: "openai/gpt-5-chat-latest", name: "GPT-5 Chat (latest)", description: "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", family: "gpt-codex", attachment: true, reasoning: true, tool_call: false, structured_output: true, temperature: true, knowledge: "2024-09-30", release_date: "2025-08-07", last_updated: "2025-08-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-5.6-sol": { id: "openai/gpt-5.6-sol", name: "GPT-5.6 Sol", description: "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", family: "gpt-sol", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2026-02-16", release_date: "2026-07-09", last_updated: "2026-07-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 64.6, metric: "resolve rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Terminal-Bench", score: 88.8, metric: "success rate", version: "2.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "DeepSWE", score: 72.7, metric: "resolve rate", version: "1.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "GPQA Diamond", score: 94.6, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "FrontierMath", score: 89, metric: "accuracy", dataset: "Tier 1-3", version: "v2", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "BrowseComp", score: 90.4, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "OSWorld", score: 62.6, metric: "success rate", version: "2.0", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "MMMU Pro", score: 83, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Agents' Last Exam", score: 52.7, source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Toolathlon", score: 58, metric: "success rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Artificial Analysis Intelligence Index", score: 58.9, metric: "index score", variant: "max", version: "4.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }, { name: "Artificial Analysis Coding Agent Index", score: 80, metric: "index score", harness: "Codex", variant: "max", version: "1.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }] }, "openai/o3-pro": { id: "openai/o3-pro", name: "o3-pro", description: "High-effort o3 tier for difficult technical reasoning and careful answers", family: "o-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05", release_date: "2025-06-10", last_updated: "2025-06-10", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 }, benchmarks: [{ name: "Aider Polyglot", score: 84.9, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-06-28" }] }, "openai/o3-deep-research": { id: "openai/o3-deep-research", name: "o3-deep-research", description: "Research model for long-horizon investigation, synthesis, and analytical reports", family: "o", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2024-05", release_date: "2024-06-26", last_updated: "2024-06-26", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 } }, "openai/gpt-5.4-pro": { id: "openai/gpt-5.4-pro", name: "GPT-5.4 Pro", description: "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", family: "gpt-pro", attachment: true, reasoning: true, tool_call: true, structured_output: false, temperature: false, knowledge: "2025-08-31", release_date: "2026-03-05", last_updated: "2026-03-05", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "GPQA Diamond", score: 94.4, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 42.7, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 58.7, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "BrowseComp", score: 89.3, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GDPval", score: 82, metric: "wins or ties", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 50, metric: "accuracy", dataset: "Tier 1-3", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 38, metric: "accuracy", dataset: "Tier 4", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "ARC-AGI-1", score: 94.5, metric: "accuracy", variant: "Verified", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "ARC-AGI-2", score: 83.3, metric: "accuracy", variant: "Verified", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FinanceAgent", score: 61.5, metric: "accuracy", version: "1.1", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GeneBench", score: 25.6, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }] }, "openai/gpt-5": { id: "openai/gpt-5", name: "GPT-5", description: "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-08-07", last_updated: "2025-08-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "Aider Polyglot", score: 88, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-08-23" }, { name: "SWE-Bench Pro", score: 41.78, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "openai/gpt-5.1-codex-mini": { id: "openai/gpt-5.1-codex-mini", name: "GPT-5.1 Codex mini", description: "Coding-optimized GPT model for repository edits, reviews, and agentic software work", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-11-13", last_updated: "2025-11-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-5-codex": { id: "openai/gpt-5-codex", name: "GPT-5-Codex", description: "Coding-optimized GPT model for repository edits, reviews, and agentic software work", family: "gpt-codex", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-09-15", last_updated: "2025-09-15", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 38.9, metric: "index", source: "https://openrouter.ai/openai/gpt-5-codex/benchmarks", date: "2026-06-01" }, { name: "SciCode", score: 40.9, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-5-codex/benchmarks", date: "2026-06-01" }, { name: "Terminal-Bench Hard", score: 37.9, metric: "success rate", source: "https://openrouter.ai/openai/gpt-5-codex/benchmarks", date: "2026-06-01" }] }, "openai/gpt-5-mini": { id: "openai/gpt-5-mini", name: "GPT-5 Mini", description: "Small GPT-5 for responsive agents, coding help, and everyday automation", family: "gpt-mini", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05-30", release_date: "2025-08-07", last_updated: "2025-08-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-5.6-luna": { id: "openai/gpt-5.6-luna", name: "GPT-5.6 Luna", description: "Cost-efficient GPT-5.6 model for fast, high-volume workloads", family: "gpt-luna", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2026-02-16", release_date: "2026-07-09", last_updated: "2026-07-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 62.7, metric: "resolve rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Terminal-Bench", score: 84.7, metric: "success rate", version: "2.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "DeepSWE", score: 67.2, metric: "resolve rate", version: "1.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "GPQA Diamond", score: 92.3, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "FrontierMath", score: 78.6, metric: "accuracy", dataset: "Tier 1-3", version: "v2", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "BrowseComp", score: 83.3, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "OSWorld", score: 45.6, metric: "success rate", version: "2.0", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "MMMU Pro", score: 78.4, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Agents' Last Exam", score: 50.3, source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Toolathlon", score: 53.4, metric: "success rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Artificial Analysis Intelligence Index", score: 51.2, metric: "index score", variant: "max", version: "4.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }, { name: "Artificial Analysis Coding Agent Index", score: 74.6, metric: "index score", harness: "Codex", variant: "max", version: "1.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }] }, "openai/gpt-5.3-codex": { id: "openai/gpt-5.3-codex", name: "GPT-5.3 Codex", description: "Coding-optimized GPT model for repository edits, reviews, and agentic software work", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2026-02-05", last_updated: "2026-02-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "SWE-Atlas Codebase QnA", score: 32.6, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 42.38, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 38.98, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }] }, "openai/gpt-oss-120b": { id: "openai/gpt-oss-120b", name: "GPT OSS 120B", description: "Open GPT reasoning model for self-hosted agents and controllable deployments", family: "gpt-oss", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2025-08-05", last_updated: "2025-08-05", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/openai/gpt-oss-120b" }] }, "openai/gpt-5.3-codex-spark": { id: "openai/gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", description: "Coding-optimized GPT model for repository edits, reviews, and agentic software work", family: "gpt-codex-spark", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2026-02-05", last_updated: "2026-02-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 128e3, input: 1e5, output: 32e3 } }, "openai/gpt-4.1-nano": { id: "openai/gpt-4.1-nano", name: "GPT-4.1 nano", description: "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", family: "gpt-nano", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-04", release_date: "2025-04-14", last_updated: "2025-04-14", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 1047576, output: 32768 }, benchmarks: [{ name: "Aider Polyglot", score: 8.9, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-04-14" }] }, "openai/o1": { id: "openai/o1", name: "o1", description: "O-series reasoning model for hard analysis, math, coding, and planning", family: "o", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2023-09", release_date: "2024-12-05", last_updated: "2024-12-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 }, benchmarks: [{ name: "Aider Polyglot", score: 61.7, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-21" }] }, "openai/gpt-5.2": { id: "openai/gpt-5.2", name: "GPT-5.2", description: "Reliable GPT generation for broad coding, writing, and tool-assisted product work", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2025-12-11", last_updated: "2025-12-11", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 29.94, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "openai/gpt-image-1": { id: "openai/gpt-image-1", name: "GPT-Image-1", description: "OpenAI image model for production generation, edits, and brand-safe visual workflows", family: "gpt-image", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2025-04-24", last_updated: "2025-04-24", modalities: { input: ["text", "image"], output: ["image"] }, open_weights: false, limit: { context: 0, output: 0 } }, "openai/gpt-realtime-2.1": { id: "openai/gpt-realtime-2.1", name: "GPT-Realtime-2.1", description: "Realtime speech-to-speech model with configurable reasoning, tool use, and robust voice-agent behavior", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: false, temperature: false, knowledge: "2024-09-30", release_date: "2026-07-06", last_updated: "2026-07-06", modalities: { input: ["text", "audio", "image"], output: ["text", "audio"] }, open_weights: false, limit: { context: 128e3, input: 96e3, output: 32e3 } }, "openai/gpt-5.4-mini": { id: "openai/gpt-5.4-mini", name: "GPT-5.4 mini", description: "Strong small GPT for coding subagents, quick tool use, and high-volume work", family: "gpt-mini", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2026-03-17", last_updated: "2026-03-17", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 54.4, metric: "resolve rate", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Terminal-Bench", score: 60, metric: "accuracy", variant: "reasoning effort xhigh", version: "2.0", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MCP Atlas", score: 57.7, metric: "score", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Toolathlon", score: 42.9, metric: "score", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "\u03C4\xB2-Bench Telecom", score: 93.4, metric: "accuracy", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "GPQA Diamond", score: 88, metric: "accuracy", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Humanity's Last Exam", score: 41.5, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Humanity's Last Exam", score: 28.2, metric: "accuracy", variant: "without tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OSWorld-Verified", score: 72.1, metric: "success rate", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MMMU Pro", score: 78, metric: "accuracy", variant: "with Python", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MMMU Pro", score: 76.6, metric: "accuracy", variant: "without tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OmniDocBench", score: 0.1263, metric: "overall edit distance", variant: "reasoning effort none", version: "1.5", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OpenAI MRCR", score: 47.7, metric: "accuracy", variant: "8-needle, 64K-128K", version: "v2", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OpenAI MRCR", score: 33.6, metric: "accuracy", variant: "8-needle, 128K-256K", version: "v2", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Graphwalks", score: 76.3, metric: "accuracy", variant: "BFS, 0-128K", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Graphwalks", score: 71.5, metric: "accuracy", variant: "parents, 0-128K", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }] }, "openai/gpt-4o-2024-05-13": { id: "openai/gpt-4o-2024-05-13", name: "GPT-4o (2024-05-13)", description: "GPT model for general reasoning, writing, coding, and tool-assisted tasks", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2023-09", release_date: "2024-05-13", last_updated: "2024-05-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 4096 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 24.2, metric: "index", source: "https://openrouter.ai/openai/gpt-4o-2024-05-13/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 30.9, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-4o-2024-05-13/benchmarks", date: "2026-03-11" }] }, "openai/o3": { id: "openai/o3", name: "o3", description: "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", family: "o", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05", release_date: "2025-04-16", last_updated: "2025-04-16", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 }, benchmarks: [{ name: "Aider Polyglot", score: 81.3, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-06-25" }] }, "openai/o4-mini-deep-research": { id: "openai/o4-mini-deep-research", name: "o4-mini-deep-research", description: "Research model for long-horizon investigation, synthesis, and analytical reports", family: "o-mini", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2024-05", release_date: "2024-06-26", last_updated: "2024-06-26", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 } }, "openai/gpt-5-pro": { id: "openai/gpt-5-pro", name: "GPT-5 Pro", description: "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", family: "gpt-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-10-06", last_updated: "2025-10-06", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 272e3 } }, "openai/gpt-5.5": { id: "openai/gpt-5.5", name: "GPT-5.5", description: "Default frontier GPT for coding, computer use, research, and knowledge work", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-12-01", release_date: "2026-04-23", last_updated: "2026-04-23", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 58.6, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "Terminal-Bench", score: 78.2, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "SWE-Atlas Codebase QnA", score: 45.43, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 44.79, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 42.59, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "Artificial Analysis Coding Agent Index", score: 65.3, metric: "average pass@1", harness: "Codex", variant: "xhigh", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 80.8, metric: "pass@1", harness: "Codex", variant: "xhigh", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 30.9, metric: "pass@1", harness: "Codex", variant: "xhigh", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 84.1, metric: "pass@1", harness: "Codex", variant: "xhigh", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Artificial Analysis Coding Agent Index", score: 60.4, metric: "average pass@1", harness: "Codex", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 79.1, metric: "pass@1", harness: "Codex", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 26.2, metric: "pass@1", harness: "Codex", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 75.8, metric: "pass@1", harness: "Codex", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Artificial Analysis Coding Agent Index", score: 57.8, metric: "average pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 75, metric: "pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 24.9, metric: "pass@1", harness: "Cursor CLI", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 73.4, metric: "pass@1", harness: "Cursor CLI", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 82.7, metric: "success rate", version: "2.0", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GPQA Diamond", score: 93.6, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 41.4, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 52.2, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "OSWorld-Verified", score: 78.7, metric: "success rate", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "BrowseComp", score: 84.4, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "MMMU Pro", score: 81.2, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "ARC-AGI-2", score: 85, metric: "accuracy", variant: "Verified", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 51.7, metric: "accuracy", dataset: "Tier 1-3", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 35.4, metric: "accuracy", dataset: "Tier 4", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GDPval", score: 84.9, metric: "wins or ties", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "MCP Atlas", score: 75.3, metric: "success rate", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Toolathlon", score: 55.6, metric: "success rate", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "\u03C4\xB2-Bench Telecom", score: 98, metric: "success rate", variant: "original prompts", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }] }, "openai/gpt-5.5-instant": { id: "openai/gpt-5.5-instant", name: "GPT-5.5 Instant", description: "Compact GPT model for low-latency assistance and high-volume workloads", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-12-01", release_date: "2026-05-05", last_updated: "2026-05-28", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 4e5, output: 128e3 } }, "openai/gpt-5.2-codex": { id: "openai/gpt-5.2-codex", name: "GPT-5.2 Codex", description: "Code-specialist GPT for repository edits, reviews, and long-running software agents", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2025-12-11", last_updated: "2025-12-11", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 41.04, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "openai/gpt-4.1": { id: "openai/gpt-4.1", name: "GPT-4.1", description: "Long-lived GPT workhorse for coding, instruction following, and production apps", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-04", release_date: "2025-04-14", last_updated: "2025-04-14", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1047576, output: 32768 }, benchmarks: [{ name: "Aider Polyglot", score: 52.4, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-04-14" }] }, "openai/gpt-4o-2024-08-06": { id: "openai/gpt-4o-2024-08-06", name: "GPT-4o (2024-08-06)", description: "GPT model for general reasoning, writing, coding, and tool-assisted tasks", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2023-09", release_date: "2024-08-06", last_updated: "2024-08-06", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 }, benchmarks: [{ name: "Aider Polyglot", score: 23.1, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-30" }, { name: "Artificial Analysis Coding Index", score: 16.6, metric: "index", source: "https://openrouter.ai/openai/gpt-4o-2024-08-06/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 33.1, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-4o-2024-08-06/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 8.3, metric: "success rate", source: "https://openrouter.ai/openai/gpt-4o-2024-08-06/benchmarks", date: "2026-03-11" }] }, "openai/gpt-5.6-terra": { id: "openai/gpt-5.6-terra", name: "GPT-5.6 Terra", description: "Balanced GPT-5.6 model for capable, cost-efficient everyday work", family: "gpt-terra", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2026-02-16", release_date: "2026-07-09", last_updated: "2026-07-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 63.4, metric: "resolve rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Terminal-Bench", score: 87.4, metric: "success rate", version: "2.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "DeepSWE", score: 69.6, metric: "resolve rate", version: "1.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "GPQA Diamond", score: 92.9, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "FrontierMath", score: 84.9, metric: "accuracy", dataset: "Tier 1-3", version: "v2", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "BrowseComp", score: 87.5, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "OSWorld", score: 50.2, metric: "success rate", version: "2.0", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "MMMU Pro", score: 80.7, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Agents' Last Exam", score: 50.4, source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Toolathlon", score: 53.1, metric: "success rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Artificial Analysis Intelligence Index", score: 55, metric: "index score", variant: "max", version: "4.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }, { name: "Artificial Analysis Coding Agent Index", score: 77.4, metric: "index score", harness: "Codex", variant: "max", version: "1.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }] }, "openai/o4-mini": { id: "openai/o4-mini", name: "o4-mini", description: "Fast o-series model for compact reasoning, coding, and tool use", family: "o-mini", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05", release_date: "2025-04-16", last_updated: "2025-04-16", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 }, benchmarks: [{ name: "Aider Polyglot", score: 72, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-04-16" }] }, "openai/gpt-5.4": { id: "openai/gpt-5.4", name: "GPT-5.4", description: "Agent-ready GPT for coding and computer-use workflows at a lower cost", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2026-03-05", last_updated: "2026-03-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 59.1, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }, { name: "SWE-Atlas Codebase QnA", score: 40.8, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Codebase QnA", score: 36.3, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 44.29, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 44.36, metric: "score", harness: "Codex CLI", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "SWE-Atlas Test Writing", score: 40, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "Artificial Analysis Coding Agent Index", score: 53.6, metric: "average pass@1", harness: "Codex", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 72.4, metric: "pass@1", harness: "Codex", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 18.4, metric: "pass@1", harness: "Codex", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 69.8, metric: "pass@1", harness: "Codex", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Artificial Analysis Coding Agent Index", score: 52.2, metric: "average pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 72.9, metric: "pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 18.9, metric: "pass@1", harness: "Cursor CLI", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 64.7, metric: "pass@1", harness: "Cursor CLI", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 75.1, metric: "success rate", version: "2.0", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GPQA Diamond", score: 92.8, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 39.8, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 52.1, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "OSWorld-Verified", score: 75, metric: "success rate", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "BrowseComp", score: 82.7, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GDPval", score: 83, metric: "wins or ties", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "ARC-AGI-2", score: 73.3, metric: "accuracy", variant: "Verified", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 47.6, metric: "accuracy", dataset: "Tier 1-3", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 27.1, metric: "accuracy", dataset: "Tier 4", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "MMMU Pro", score: 81.2, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }] }, "openai/o3-mini": { id: "openai/o3-mini", name: "o3-mini", description: "Smaller o-series reasoner for economical coding, math, and planning tasks", family: "o-mini", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05", release_date: "2024-12-20", last_updated: "2025-01-29", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 }, benchmarks: [{ name: "Aider Polyglot", score: 60.4, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-01-31" }] }, "openai/gpt-4o-2024-11-20": { id: "openai/gpt-4o-2024-11-20", name: "GPT-4o (2024-11-20)", description: "GPT model for general reasoning, writing, coding, and tool-assisted tasks", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2023-09", release_date: "2024-11-20", last_updated: "2024-11-20", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 }, benchmarks: [{ name: "Aider Polyglot", score: 18.2, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-30" }, { name: "Artificial Analysis Coding Index", score: 16.7, metric: "index", source: "https://openrouter.ai/openai/gpt-4o-2024-11-20/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 33.3, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-4o-2024-11-20/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 8.3, metric: "success rate", source: "https://openrouter.ai/openai/gpt-4o-2024-11-20/benchmarks", date: "2026-03-11" }] }, "openai/gpt-oss-safeguard-120b": { id: "openai/gpt-oss-safeguard-120b", name: "GPT OSS Safeguard 120B", description: "Safety model for policy screening, moderation, and risk-aware routing workflows", family: "gpt-oss", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2025-10-29", last_updated: "2025-10-29", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/openai/gpt-oss-safeguard-120b" }] }, "openai/gpt-realtime-whisper": { id: "openai/gpt-realtime-whisper", name: "GPT Realtime Whisper", description: "Streaming speech-to-text model for low-latency transcript deltas from live audio", family: "whisper", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2026-05-07", last_updated: "2026-05-07", modalities: { input: ["audio"], output: ["text"] }, open_weights: false, limit: { context: 0, output: 0 } }, "openai/gpt-5.2-chat-latest": { id: "openai/gpt-5.2-chat-latest", name: "GPT-5.2 Chat", description: "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2025-12-11", last_updated: "2025-12-11", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 } }, "openai/gpt-5.2-pro": { id: "openai/gpt-5.2-pro", name: "GPT-5.2 Pro", description: "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows", family: "gpt-pro", attachment: true, reasoning: true, tool_call: true, structured_output: false, temperature: false, knowledge: "2025-08-31", release_date: "2025-12-11", last_updated: "2025-12-11", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-5.1-chat-latest": { id: "openai/gpt-5.1-chat-latest", name: "GPT-5.1 Chat", description: "Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-11-13", last_updated: "2025-11-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 } }, "openai/gpt-image-2": { id: "openai/gpt-image-2", name: "GPT-Image-2", description: "Image model for prompt-driven generation, editing, and visual design workflows", family: "gpt-image", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-04-21", last_updated: "2026-04-21", modalities: { input: ["text", "image"], output: ["image"] }, open_weights: false, limit: { context: 0, output: 0 } }, "openai/gpt-5.1-codex-max": { id: "openai/gpt-5.1-codex-max", name: "GPT-5.1 Codex Max", description: "Coding-optimized GPT model for repository edits, reviews, and agentic software work", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-11-13", last_updated: "2025-11-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-5.1-codex": { id: "openai/gpt-5.1-codex", name: "GPT-5.1 Codex", description: "Codex GPT for repository edits, code review, and practical software agents", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-11-13", last_updated: "2025-11-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-4.1-mini": { id: "openai/gpt-4.1-mini", name: "GPT-4.1 mini", description: "Affordable GPT-4.1 lane for fast coding help and structured extraction", family: "gpt-mini", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-04", release_date: "2025-04-14", last_updated: "2025-04-14", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1047576, output: 32768 }, benchmarks: [{ name: "Aider Polyglot", score: 32.4, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-04-14" }] }, "openai/gpt-oss-20b": { id: "openai/gpt-oss-20b", name: "GPT OSS 20B", description: "Open GPT reasoning model for self-hosted agents and controllable deployments", family: "gpt-oss", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2025-08-05", last_updated: "2025-08-05", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/openai/gpt-oss-20b" }] }, "openai/whisper-large-v3-turbo": { id: "openai/whisper-large-v3-turbo", name: "Whisper Large v3 Turbo", description: "Speech transcription model for accurate audio-to-text and captioning workflows", family: "whisper", attachment: false, reasoning: false, tool_call: false, release_date: "2024-10-01", last_updated: "2024-10-01", modalities: { input: ["audio"], output: ["text"] }, open_weights: true, limit: { context: 448, output: 448 } }, "openai/gpt-5.4-nano": { id: "openai/gpt-5.4-nano", name: "GPT-5.4 nano", description: "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", family: "gpt-nano", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2026-03-17", last_updated: "2026-03-17", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 52.4, metric: "resolve rate", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Terminal-Bench", score: 46.3, metric: "accuracy", variant: "reasoning effort xhigh", version: "2.0", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MCP Atlas", score: 56.1, metric: "score", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Toolathlon", score: 35.5, metric: "score", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "\u03C4\xB2-Bench Telecom", score: 92.5, metric: "accuracy", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "GPQA Diamond", score: 82.8, metric: "accuracy", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Humanity's Last Exam", score: 37.7, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Humanity's Last Exam", score: 24.3, metric: "accuracy", variant: "without tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OSWorld-Verified", score: 39, metric: "success rate", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MMMU Pro", score: 69.5, metric: "accuracy", variant: "with Python", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MMMU Pro", score: 66.1, metric: "accuracy", variant: "without tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OmniDocBench", score: 0.2419, metric: "overall edit distance", variant: "reasoning effort none", version: "1.5", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OpenAI MRCR", score: 44.2, metric: "accuracy", variant: "8-needle, 64K-128K", version: "v2", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OpenAI MRCR", score: 33.1, metric: "accuracy", variant: "8-needle, 128K-256K", version: "v2", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Graphwalks", score: 73.4, metric: "accuracy", variant: "BFS, 0-128K", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Graphwalks", score: 50.8, metric: "accuracy", variant: "parents, 0-128K", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }] }, "openai/gpt-5.1": { id: "openai/gpt-5.1", name: "GPT-5.1", description: "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-11-13", last_updated: "2025-11-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "xai/grok-4.6": { id: "xai/grok-4.6", name: "Grok 4.6", description: "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects", family: "grok", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-02-01", release_date: "2026-08-12", last_updated: "2026-08-12", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 5e5, output: 5e5 } }, "xai/grok-4.5": { id: "xai/grok-4.5", name: "Grok 4.5", description: "xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk", family: "grok", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-07-08", last_updated: "2026-07-08", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 5e5, output: 5e5 }, benchmarks: [{ name: "SWE-Bench Pro", score: 64.7, metric: "resolve rate", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }, { name: "SWE-Bench Multilingual", score: 78, metric: "resolve rate", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }, { name: "Terminal-Bench", score: 83.3, metric: "success rate", version: "2.1", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }, { name: "DeepSWE", score: 62, metric: "resolve rate", version: "1.0", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }, { name: "DeepSWE", score: 53, metric: "resolve rate", harness: "mini-swe-agent", version: "1.1", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }, { name: "SWE Marathon", score: 29, metric: "pass@1", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }] }, "xai/grok-4.20-0309-non-reasoning": { id: "xai/grok-4.20-0309-non-reasoning", name: "Grok 4.20 (Non-Reasoning)", description: "Grok model for agentic tool use, reasoning, coding, and live assistance", family: "grok", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, release_date: "2026-03-09", last_updated: "2026-03-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 3e4 } }, "xai/grok-imagine-video-1.5": { id: "xai/grok-imagine-video-1.5", name: "Grok Imagine Video 1.5", description: "Video model for image-to-video generation, editing, and extension workflows", family: "grok", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-05-30", last_updated: "2026-05-30", modalities: { input: ["text", "image", "video"], output: ["video"] }, open_weights: false, limit: { context: 1024, output: 0 } }, "xai/grok-4.1-fast": { id: "xai/grok-4.1-fast", name: "Grok 4.1 Fast", description: "xAI's fast agentic tool-calling model with a 2M context window; non-reasoning variant for low-latency responses", family: "grok", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, release_date: "2025-11-19", last_updated: "2025-11-19", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e6, output: 3e4 } }, "xai/grok-4.20-0309-reasoning": { id: "xai/grok-4.20-0309-reasoning", name: "Grok 4.20 (Reasoning)", description: "Reasoning Grok for document-heavy analysis and long-horizon tool use", family: "grok", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-03-09", last_updated: "2026-03-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 3e4 } }, "xai/grok-imagine-image-2.0": { id: "xai/grok-imagine-image-2.0", name: "Grok Imagine Image 2.0", description: "Image model for prompt-driven generation, editing, and visual design workflows", family: "grok", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-08-07", last_updated: "2026-08-07", modalities: { input: ["text", "image"], output: ["image"] }, open_weights: false, limit: { context: 8e3, output: 0 } }, "xai/grok-build-0.1": { id: "xai/grok-build-0.1", name: "Grok Build 0.1", description: "Fast Grok coding model tuned for agentic engineering and iterative edits", family: "grok-build", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-16", last_updated: "2026-04-16", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 256e3 } }, "xai/grok-4.3": { id: "xai/grok-4.3", name: "Grok 4.3", description: "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", family: "grok", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-17", last_updated: "2026-04-17", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 3e4 }, benchmarks: [{ name: "Artificial Analysis Intelligence Index", score: 53, metric: "index score", version: "4.0", source: "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing", date: "2026-04-30" }, { name: "GDPval-AA", score: 1500, metric: "Elo", source: "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing", date: "2026-04-30" }, { name: "\u03C4\xB2-Bench Telecom", score: 98, metric: "success rate", source: "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing", date: "2026-04-30" }, { name: "IFBench", score: 81, metric: "accuracy", source: "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing", date: "2026-04-30" }] }, "meituan/longcat-2.0": { id: "meituan/longcat-2.0", name: "LongCat-2.0", description: "Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window", family: "longcat", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-30", last_updated: "2026-06-30", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 131072 }, benchmarks: [{ name: "SWE-Bench Pro", score: 59.5, metric: "resolve rate", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "SWE-Bench Multilingual", score: 77.3, metric: "resolve rate", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "Terminal-Bench", score: 70.8, metric: "success rate", version: "2.1", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "GPQA Diamond", score: 88.9, metric: "accuracy", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "BrowseComp", score: 79.9, metric: "accuracy", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "IFEval", score: 90, metric: "accuracy", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "FORTE", score: 73.2, metric: "success rate", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }] }, "swiss-ai/apertus-70b": { id: "swiss-ai/apertus-70b", name: "Apertus 70B", description: "Fully open 70B multilingual LLM supporting 1800+ languages with 65K context. Trained on 15T tokens of compliant open data. Apache 2.0, EU AI Act compliant.", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-09", release_date: "2025-09-02", last_updated: "2025-09-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 65536, output: 8192 }, license: "Apache-2.0", links: [{ label: "Paper", url: "https://arxiv.org/abs/2509.14233", type: "paper" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/swiss-ai/Apertus-70B-Instruct-2509" }] }, "swiss-ai/apertus-8b": { id: "swiss-ai/apertus-8b", name: "Apertus 8B", description: "Fully open 8B multilingual LLM supporting 1800+ languages with 65K context. Trained on compliant open data. Apache 2.0, EU AI Act compliant.", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-09", release_date: "2025-09-02", last_updated: "2025-09-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 65536, output: 8192 }, license: "Apache-2.0", links: [{ label: "Paper", url: "https://arxiv.org/abs/2509.14233", type: "paper" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/swiss-ai/Apertus-8B-Instruct-2509" }] }, "stepfun/step-3.5-flash-2603": { id: "stepfun/step-3.5-flash-2603", name: "Step 3.5 Flash 2603", description: "StepFun flash model for efficient multimodal reasoning, coding, and tool use", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-01", release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, input: 256e3, output: 256e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/stepfun-ai/Step-3.5-Flash" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 34.6, metric: "index", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }, { name: "SciCode", score: 38.5, metric: "percent correct", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }, { name: "Terminal-Bench Hard", score: 32.6, metric: "success rate", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }] }, "stepfun/step-3.5-flash": { id: "stepfun/step-3.5-flash", name: "Step 3.5 Flash", description: "StepFun flash lane for quick multimodal reasoning and coding assistance", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-01", release_date: "2026-01-29", last_updated: "2026-02-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, input: 256e3, output: 256e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/stepfun-ai/Step-3.5-Flash" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 31.6, metric: "index", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }, { name: "SciCode", score: 40.4, metric: "percent correct", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }, { name: "Terminal-Bench Hard", score: 27.3, metric: "success rate", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }, { name: "SWE-Bench Verified", score: 74.4, metric: "resolved", source: "https://arxiv.org/abs/2602.10604" }] }, "stepfun/step-3.7-flash": { id: "stepfun/step-3.7-flash", name: "Step 3.7 Flash", description: "Newer StepFun flash model for faster agents, coding, and multimodal prompts", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2026-03-01", release_date: "2026-05-29", last_updated: "2026-05-29", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 256e3, input: 256e3, output: 256e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/stepfun-ai/Step-3.7-Flash" }], benchmarks: [{ name: "SWE-Bench Pro", score: 56.3, metric: "resolve rate", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "SWE-Bench Verified", score: 76.5, metric: "resolved", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "Terminal-Bench", score: 59.6, metric: "success rate", version: "2.1", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "Humanity's Last Exam", score: 47.2, metric: "accuracy", variant: "with tools", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "BrowseComp", score: 75.8, metric: "accuracy", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "Toolathlon", score: 49.5, metric: "success rate", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "GDPval", score: 45.8, metric: "wins or ties", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "ClawEval", score: 67.1, metric: "pass^3", version: "1.1", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "Artificial Analysis Coding Index", score: 37.1, metric: "index", source: "https://openrouter.ai/stepfun/step-3.7-flash/benchmarks", date: "2026-06-15" }, { name: "SciCode", score: 40, metric: "percent correct", source: "https://openrouter.ai/stepfun/step-3.7-flash/benchmarks", date: "2026-06-15" }, { name: "Terminal-Bench Hard", score: 35.6, metric: "success rate", source: "https://openrouter.ai/stepfun/step-3.7-flash/benchmarks", date: "2026-06-15" }] }, "sarvam/sarvam-105b": { id: "sarvam/sarvam-105b", name: "Sarvam 105B", description: "Flagship Indian-language reasoning model for enterprise multilingual applications", family: "sarvam", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-09-01", last_updated: "2025-09-01", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 } }, "sarvam/sarvam-30b": { id: "sarvam/sarvam-30b", name: "Sarvam 30B", description: "Efficient Indian-language reasoning model for chat, coding, and multilingual work", family: "sarvam", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-18", last_updated: "2026-02-18", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 } }, "trendyol/asure-12b": { id: "trendyol/asure-12b", name: "Trendyol Asure 12B", description: "Turkish-language multimodal instruct model built on Gemma 3 12B for e-commerce text, chat, and image-text tasks", family: "gemma", attachment: true, reasoning: false, tool_call: false, temperature: true, release_date: "2026-02-19", last_updated: "2026-02-20", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 131072 }, license: "Gemma", weights: [{ label: "Hugging Face", url: "https://huggingface.co/Trendyol/Trendyol-LLM-Asure-12B" }] }, "deepreinforce/ornith-1.0-35b": { id: "deepreinforce/ornith-1.0-35b", name: "Ornith 1.0 35B", description: "Large coding-reasoning model for agentic software tasks and RL search", family: "ornith", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-25", last_updated: "2026-06-25", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144 }, license: "MIT", links: [{ label: "Model card", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B", type: "model_card" }, { label: "Announcement", url: "https://deep-reinforce.com/ornith_1_0.html", type: "announcement" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 75.6, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "SWE-Bench Pro", score: 50.4, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "SWE-Bench Multilingual", score: 69.3, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "Terminal-Bench 2.1", score: 64.2, metric: "percent", variant: "Terminus-2", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "Terminal-Bench 2.1", score: 62.8, metric: "percent", variant: "Claude Code", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "NL2Repo", score: 34.6, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "Claw-eval", score: 69.8, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }] }, "deepreinforce/ornith-1.0-397b": { id: "deepreinforce/ornith-1.0-397b", name: "Ornith 1.0 397B", description: "Large coding-reasoning model for agentic software tasks and RL search", family: "ornith", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-25", last_updated: "2026-06-25", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144 }, license: "MIT", links: [{ label: "Model card", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B", type: "model_card" }, { label: "Announcement", url: "https://deep-reinforce.com/ornith_1_0.html", type: "announcement" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { label: "Hugging Face (FP8)", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B-FP8", quantization: "fp8" }], benchmarks: [{ name: "SWE-Bench Verified", score: 82.4, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "SWE-Bench Pro", score: 62.2, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "SWE-Bench Multilingual", score: 78.9, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "Terminal-Bench 2.1", score: 77.5, metric: "percent", variant: "Terminus-2", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "Terminal-Bench 2.1", score: 78.2, metric: "percent", variant: "Claude Code", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "NL2Repo", score: 48.2, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "Claw-eval", score: 77.1, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }] }, "deepreinforce/ornith-1.0-31b": { id: "deepreinforce/ornith-1.0-31b", name: "Ornith 1.0 31B", description: "Open coding-reasoning model for repository tasks and self-improving agents", family: "ornith", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-25", last_updated: "2026-06-25", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144 }, license: "MIT", links: [{ label: "Announcement", url: "https://deep-reinforce.com/ornith_1_0.html", type: "announcement" }] }, "deepreinforce/ornith-1.0-9b": { id: "deepreinforce/ornith-1.0-9b", name: "Ornith 1.0 9B", description: "Open coding-reasoning model for repository tasks and self-improving agents", family: "ornith", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-25", last_updated: "2026-06-25", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144 }, license: "MIT", links: [{ label: "Model card", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B", type: "model_card" }, { label: "Announcement", url: "https://deep-reinforce.com/ornith_1_0.html", type: "announcement" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 69.4, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "SWE-Bench Pro", score: 42.9, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "SWE-Bench Multilingual", score: 52, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "Terminal-Bench 2.1", score: 43.1, metric: "percent", variant: "Terminus-2", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "Terminal-Bench 2.1", score: 40.6, metric: "percent", variant: "Claude Code", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "NL2Repo", score: 27.2, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "Claw-eval", score: 63.1, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }] }, "xiaomi/mimo-v2.5-pro-ultraspeed": { id: "xiaomi/mimo-v2.5-pro-ultraspeed", name: "MiMo-V2.5-Pro-UltraSpeed", description: "MiMo pro model for strong multimodal reasoning and agent execution", family: "mimo", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12", release_date: "2026-06-08", last_updated: "2026-06-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro-FP4-DFlash" }] }, "xiaomi/mimo-v2.5": { id: "xiaomi/mimo-v2.5", name: "MiMo-V2.5", description: "Open MiMo model for multimodal coding agents and long-context automation", family: "mimo", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12", release_date: "2026-04-22", last_updated: "2026-04-22", modalities: { input: ["text", "image", "audio", "video"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/XiaomiMiMo/MiMo-V2.5" }] }, "xiaomi/mimo-v2-pro": { id: "xiaomi/mimo-v2-pro", name: "MiMo-V2-Pro", description: "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", family: "mimo", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12", release_date: "2026-03-18", last_updated: "2026-03-18", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 131072 } }, "xiaomi/mimo-v2-omni": { id: "xiaomi/mimo-v2-omni", name: "MiMo-V2-Omni", description: "MiMo omni model for text, image, video, audio, and agents", family: "mimo", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12", release_date: "2026-03-18", last_updated: "2026-03-18", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 131072 } }, "xiaomi/mimo-v2-flash": { id: "xiaomi/mimo-v2-flash", name: "MiMo-V2-Flash", description: "MiMo flash model for fast multimodal assistance and agent workflows", family: "mimo", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12-01", release_date: "2025-12-16", last_updated: "2026-02-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/XiaomiMiMo/MiMo-V2-Flash" }] }, "xiaomi/mimo-v2.5-pro": { id: "xiaomi/mimo-v2.5-pro", name: "MiMo-V2.5-Pro", description: "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", family: "mimo", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12", release_date: "2026-04-22", last_updated: "2026-04-22", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro" }], benchmarks: [{ name: "SWE-Bench Verified", score: 78.9, metric: "resolved", source: "https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro" }, { name: "SWE-Bench Pro", score: 57.2, metric: "resolve rate", source: "https://mimo.xiaomi.com/mimo-v2-5-pro/", date: "2026-04-22" }, { name: "GPQA Diamond", score: 86.6, metric: "accuracy", source: "https://mimo.xiaomi.com/mimo-v2-5-pro/", date: "2026-04-22" }] }, "alibaba/qwen3-vl-plus": { id: "alibaba/qwen3-vl-plus", name: "Qwen3-VL Plus", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-09-23", last_updated: "2025-09-23", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 32768 } }, "alibaba/qwen3.8-max-preview": { id: "alibaba/qwen3.8-max-preview", name: "Qwen3.8 Max Preview", description: "Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows", family: "qwen", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-07-19", last_updated: "2026-07-19", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 131072 }, benchmarks: [{ name: "Terminal-Bench", score: 86.6, metric: "accuracy", variant: "xhigh", version: "2.1", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "SWE-Bench Pro", score: 67.7, metric: "resolve rate", harness: "Claude Code", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "DeepSWE", score: 56.6, metric: "resolve rate", harness: "Claude Code", variant: "xhigh", version: "1.1", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "NL2Repo", score: 55.9, metric: "resolve rate", harness: "Claude Code", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "FrontierSWE", score: 73.5, metric: "dominance score", harness: "Claude Code", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "MLS-Bench-Lite", score: 41, metric: "score", harness: "Claude Code", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "AutomationBench", score: 27.3, metric: "pass@1", variant: "xhigh", dataset: "600-task public subset", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "Toolathlon Verified", score: 72.5, metric: "pass@1", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "WideSearch", score: 81.9, metric: "F1", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "Humanity's Last Exam", score: 56.2, metric: "accuracy", variant: "xhigh, with tools", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "GPQA Diamond", score: 92.6, metric: "accuracy", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "Humanity's Last Exam", score: 43.6, metric: "accuracy", variant: "xhigh, no tools", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "IFBench", score: 82.8, metric: "score", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "OSWorld-Verified", score: 86.1, metric: "success rate", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "MMMU Pro", score: 82.3, metric: "accuracy", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }] }, "alibaba/qwen3-coder-30b-a3b-instruct": { id: "alibaba/qwen3-coder-30b-a3b-instruct", name: "Qwen3-Coder 30B-A3B Instruct", description: "Smaller Qwen coder for efficient local agents and repo-level fixes", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-04", last_updated: "2025-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-Coder-30B-A3B-Instruct" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 19.4, metric: "index", source: "https://openrouter.ai/qwen/qwen3-coder-30b-a3b-instruct/benchmarks", date: "2026-06-02" }, { name: "SciCode", score: 27.8, metric: "percent correct", source: "https://openrouter.ai/qwen/qwen3-coder-30b-a3b-instruct/benchmarks", date: "2026-06-02" }, { name: "Terminal-Bench Hard", score: 15.2, metric: "success rate", source: "https://openrouter.ai/qwen/qwen3-coder-30b-a3b-instruct/benchmarks", date: "2026-06-02" }] }, "alibaba/qwen3.7-max": { id: "alibaba/qwen3.7-max", name: "Qwen3.7 Max", description: "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-05-21", last_updated: "2026-05-21", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 }, benchmarks: [{ name: "SWE-Bench Verified", score: 80.4, metric: "resolved", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "SWE-Bench Pro", score: 60.6, metric: "resolve rate", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "SWE-Bench Multilingual", score: 78.3, metric: "resolve rate", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "Terminal-Bench", score: 69.7, metric: "success rate", harness: "Terminus-2", version: "2.0", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "GPQA Diamond", score: 92.4, metric: "accuracy", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "Humanity's Last Exam", score: 41.4, metric: "accuracy", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "SciCode", score: 53.5, source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "MCP Atlas", score: 76.4, metric: "success rate", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "NL2Repo", score: 47.2, harness: "Claude Code", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }] }, "alibaba/qwen-turbo": { id: "alibaba/qwen-turbo", name: "Qwen Turbo", description: "Efficient Qwen model for fast chat, extraction, and high-volume workloads", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-11-01", last_updated: "2025-04-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 16384 } }, "alibaba/qwen-omni-turbo": { id: "alibaba/qwen-omni-turbo", name: "Qwen-Omni Turbo", description: "Qwen omni model for text, vision, audio, and multimodal agent tasks", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2025-01-19", last_updated: "2025-03-26", modalities: { input: ["text", "image", "audio", "video"], output: ["text", "audio"] }, open_weights: false, limit: { context: 32768, output: 2048 } }, "alibaba/qwen-vl-max": { id: "alibaba/qwen-vl-max", name: "Qwen-VL Max", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-04-08", last_updated: "2025-08-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 8192 } }, "alibaba/qwen3-32b": { id: "alibaba/qwen3-32b", name: "Qwen3 32B", description: "Dense open Qwen model for self-hosted chat, reasoning, and coding", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-04", last_updated: "2025-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-32B" }], benchmarks: [{ name: "Aider Polyglot", score: 40, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-08" }] }, "alibaba/qwen3-235b-a22b": { id: "alibaba/qwen3-235b-a22b", name: "Qwen3 235B-A22B", description: "Large open Qwen MoE for multilingual reasoning, coding, and tool use", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-04", last_updated: "2025-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-235B-A22B" }], benchmarks: [{ name: "Aider Polyglot", score: 59.6, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-09" }, { name: "SWE-Bench Pro", score: 21.41, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "alibaba/qwen-max": { id: "alibaba/qwen-max", name: "Qwen Max", description: "Flagship Qwen model for complex reasoning, coding, and agentic workflows", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-04-03", last_updated: "2025-01-25", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 32768, output: 8192 }, benchmarks: [{ name: "Aider Polyglot", score: 21.8, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-01-28" }] }, "alibaba/qwen3.5-35b-a3b": { id: "alibaba/qwen3.5-35b-a3b", name: "Qwen3.5 35B-A3B", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-23", last_updated: "2026-02-23", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.5-35B-A3B" }] }, "alibaba/qwen3-vl-235b-a22b-thinking": { id: "alibaba/qwen3-vl-235b-a22b-thinking", name: "Qwen3 VL 235B A22B Thinking", description: "Qwen vision-language thinking model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-09-23", last_updated: "2025-09-23", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-VL-235B-A22B-Thinking" }] }, "alibaba/qwen3-30b-a3b": { id: "alibaba/qwen3-30b-a3b", name: "Qwen3 30B A3B", description: "Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-04-28", last_updated: "2025-04-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-30B-A3B" }] }, "alibaba/qwen3.5-397b-a17b": { id: "alibaba/qwen3.5-397b-a17b", name: "Qwen3.5 397B-A17B", description: "Large open Qwen multimodal MoE for visual agents and long technical tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-15", last_updated: "2026-02-15", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.5-397B-A17B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 76.4, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.5-397B-A17B" }] }, "alibaba/qwen3.5-flash": { id: "alibaba/qwen3.5-flash", name: "Qwen3.5 Flash", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-23", last_updated: "2026-02-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 } }, "alibaba/qwen2.5-coder-0.5b": { id: "alibaba/qwen2.5-coder-0.5b", name: "Qwen2.5-Coder-0.5B", description: "Tiny open Qwen code model for lightweight completion and on-device coding", family: "qwen", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2024-11-12", last_updated: "2024-11-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 32768, output: 8192 }, license: "Apache 2.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen2.5-Coder-0.5B" }] }, "alibaba/qwen-plus": { id: "alibaba/qwen-plus", name: "Qwen Plus", description: "Qwen instruction model for multilingual chat, reasoning, and tool use", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-01-25", last_updated: "2025-09-11", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 32768 } }, "alibaba/qwen3-coder-next": { id: "alibaba/qwen3-coder-next", name: "Qwen3 Coder Next", description: "Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use", family: "qwen", attachment: false, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-09", release_date: "2026-02-03", last_updated: "2026-02-03", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-Coder-Next" }] }, "alibaba/qwen3.5-plus": { id: "alibaba/qwen3.5-plus", name: "Qwen3.5 Plus", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-02-16", last_updated: "2026-02-16", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 } }, "alibaba/qwen2-5-vl-72b-instruct": { id: "alibaba/qwen2-5-vl-72b-instruct", name: "Qwen2.5-VL 72B Instruct", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-09", last_updated: "2024-09", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen2.5-VL-72B-Instruct" }] }, "alibaba/qwen3-coder-flash": { id: "alibaba/qwen3-coder-flash", name: "Qwen3 Coder Flash", description: "Qwen coding model for software agents, repository edits, and code reasoning", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-07-28", last_updated: "2025-07-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 } }, "alibaba/qwen3.6-max-preview": { id: "alibaba/qwen3.6-max-preview", name: "Qwen3.6 Max Preview", description: "Flagship Qwen model for complex reasoning, coding, and agentic workflows", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-04-20", last_updated: "2026-04-20", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 65536 } }, "alibaba/qwen3.8-max": { id: "alibaba/qwen3.8-max", name: "Qwen3.8 Max", description: "2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows", family: "qwen", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-08-03", last_updated: "2026-08-03", modalities: { input: ["text", "image", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 131072 } }, "alibaba/qwen3.7-plus": { id: "alibaba/qwen3.7-plus", name: "Qwen3.7 Plus", description: "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", family: "qwen", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-06-02", last_updated: "2026-06-02", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 64e3 } }, "alibaba/qwq-32b": { id: "alibaba/qwq-32b", name: "QwQ 32B", description: "Open reasoning model from the Qwen team for math, coding, and step-by-step problem solving", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2025-03-05", last_updated: "2025-03-05", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/QwQ-32B" }] }, "alibaba/qwen3.7-flash": { id: "alibaba/qwen3.7-flash", name: "Qwen3.7 Flash", description: "Lightweight multimodal Qwen model for high-throughput text, image, and video tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-07-15", last_updated: "2026-07-15", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, input: 991e3, output: 65536 } }, "alibaba/qwen3-next-80b-a3b-thinking": { id: "alibaba/qwen3-next-80b-a3b-thinking", name: "Qwen3-Next 80B-A3B (Thinking)", description: "Efficient Qwen thinking model for local reasoning, math, and coding agents", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-09", last_updated: "2025-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Thinking" }] }, "alibaba/qwen3-coder-480b-a35b-instruct": { id: "alibaba/qwen3-coder-480b-a35b-instruct", name: "Qwen3-Coder 480B-A35B Instruct", description: "Open Qwen coding heavyweight for repository reasoning and agentic engineering", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-04", last_updated: "2025-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-Coder-480B-A35B-Instruct" }], benchmarks: [{ name: "SWE-Bench Pro", score: 38.7, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "alibaba/qwen2.5-coder-32b-instruct": { id: "alibaba/qwen2.5-coder-32b-instruct", name: "Qwen2.5-Coder-32B-Instruct", description: "Open coding-focused Qwen model for code generation, repair, and repository reasoning", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2024-11-12", last_updated: "2024-11-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen2.5-Coder-32B-Instruct" }] }, "alibaba/qwen-flash": { id: "alibaba/qwen-flash", name: "Qwen Flash", description: "Efficient Qwen model for fast chat, extraction, and high-volume workloads", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2025-07-28", last_updated: "2025-07-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 32768 } }, "alibaba/qwen3-max": { id: "alibaba/qwen3-max", name: "Qwen3 Max", description: "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-09-23", last_updated: "2025-09-23", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 65536 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 26.4, metric: "index", source: "https://openrouter.ai/qwen/qwen3-max/benchmarks", date: "2026-05-30" }, { name: "SciCode", score: 38.3, metric: "percent correct", source: "https://openrouter.ai/qwen/qwen3-max/benchmarks", date: "2026-05-30" }, { name: "Terminal-Bench Hard", score: 20.5, metric: "success rate", source: "https://openrouter.ai/qwen/qwen3-max/benchmarks", date: "2026-05-30" }] }, "alibaba/qwen3.6-35b-a3b": { id: "alibaba/qwen3.6-35b-a3b", name: "Qwen3.6 35B-A3B", description: "Open multimodal Qwen MoE for local agents that need vision, audio, and code", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-17", last_updated: "2026-04-17", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.6-35B-A3B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 73.4, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.6-35B-A3B" }] }, "alibaba/qwen3-235b-a22b-instruct-2507": { id: "alibaba/qwen3-235b-a22b-instruct-2507", name: "Qwen3 235B-A22B Instruct 2507", description: "Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-07-21", last_updated: "2025-07-21", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 16384 }, license: "Apache 2.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507" }] }, "alibaba/qwen3.6-27b": { id: "alibaba/qwen3.6-27b", name: "Qwen3.6 27B", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-22", last_updated: "2026-04-22", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.6-27B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 77.2, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.6-27B" }] }, "alibaba/qwen3-next-80b-a3b-instruct": { id: "alibaba/qwen3-next-80b-a3b-instruct", name: "Qwen3-Next 80B-A3B Instruct", description: "Qwen instruction model for multilingual chat, reasoning, and tool use", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-09", last_updated: "2025-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Instruct" }] }, "alibaba/qwen3.8-27b": { id: "alibaba/qwen3.8-27b", name: "Qwen3.8 27B", description: "Dense 27B vision-language model for coding, agent tasks, and image and video understanding", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-14", last_updated: "2026-08-14", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.8-27B" }], benchmarks: [{ name: "SWE-bench Pro", score: 61.7, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.8-27B" }] }, "alibaba/qwen3.6-plus": { id: "alibaba/qwen3.6-plus", name: "Qwen3.6 Plus", description: "Earlier Qwen multimodal workhorse for million-token agent and document tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 } }, "alibaba/qwen3.5-9b": { id: "alibaba/qwen3.5-9b", name: "Qwen3.5 9B", description: "Qwen instruction model for multilingual chat, reasoning, and tool use", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-23", last_updated: "2026-02-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.5-9B" }] }, "alibaba/qwen3.5-27b": { id: "alibaba/qwen3.5-27b", name: "Qwen3.5 27B", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-23", last_updated: "2026-02-23", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.5-27B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 72.4, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.5-27B" }] }, "alibaba/qwq-plus": { id: "alibaba/qwq-plus", name: "QwQ Plus", description: "Qwen reasoning model for deliberate problem solving, math, and coding", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2025-03-05", last_updated: "2025-03-05", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 8192 } }, "alibaba/qwen3.5-122b-a10b": { id: "alibaba/qwen3.5-122b-a10b", name: "Qwen3.5 122B-A10B", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-23", last_updated: "2026-02-23", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.5-122B-A10B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 72, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.5-122B-A10B" }] }, "alibaba/qwen3-coder-plus": { id: "alibaba/qwen3-coder-plus", name: "Qwen3 Coder Plus", description: "Hosted Qwen coder for software agents, repo edits, and long-context code", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-07-23", last_updated: "2025-07-23", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "alibaba/qwen3-vl-235b-a22b-instruct": { id: "alibaba/qwen3-vl-235b-a22b-instruct", name: "Qwen3 VL 235B A22B Instruct", description: "Qwen vision-language instruct model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-09-23", last_updated: "2025-09-23", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-VL-235B-A22B-Instruct" }] }, "alibaba/qwen-vl-plus": { id: "alibaba/qwen-vl-plus", name: "Qwen-VL Plus", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-01-25", last_updated: "2025-08-15", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 8192 } }, "alibaba/qwen3.6-flash": { id: "alibaba/qwen3.6-flash", name: "Qwen3.6 Flash", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen3.6", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-27", last_updated: "2026-04-27", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 } }, "alibaba/qwen3.8-2.4t-a95b": { id: "alibaba/qwen3.8-2.4t-a95b", name: "Qwen3.8 2.4T A95B", description: "Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows", family: "qwen", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-12", last_updated: "2026-08-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 131072 }, license: "qwen3.8-max", weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B" }] }, "sakana/fugu": { id: "sakana/fugu", name: "Fugu", description: "Multi-agent model for routing expert agents across complex analytical tasks", family: "fugu", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, release_date: "2026-06-15", last_updated: "2026-06-15", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 1e6 }, links: [{ label: "Official model catalog", url: "https://raw.githubusercontent.com/SakanaAI/fugu/refs/heads/main/configs/files/fugu.json", type: "docs" }], benchmarks: [{ name: "SWE Bench Pro", score: 59, source: "https://console.sakana.ai/models" }, { name: "Terminal Bench 2.1", score: 80.2, source: "https://console.sakana.ai/models" }, { name: "LiveCodeBench", score: 92.9, source: "https://console.sakana.ai/models" }, { name: "LiveCodeBench Pro", score: 87.8, source: "https://console.sakana.ai/models" }, { name: "Humanity\u2019s Last Exam", score: 47.2, source: "https://console.sakana.ai/models" }, { name: "CharXiv Reasoning", score: 85.1, source: "https://console.sakana.ai/models" }, { name: "GPQA Diamond", score: 95.5, source: "https://console.sakana.ai/models" }, { name: "SciCode", score: 60.1, source: "https://console.sakana.ai/models" }, { name: "\u03C43 Banking", score: 21.7, source: "https://console.sakana.ai/models" }, { name: "Long Context Reasoning", score: 74.7, source: "https://console.sakana.ai/models" }, { name: "MRCRv2", score: 86.6, source: "https://console.sakana.ai/models" }, { name: "CTI-REALM", score: 67.5, source: "https://console.sakana.ai/models" }] }, "sakana/sakana-namazu": { id: "sakana/sakana-namazu", name: "Sakana Namazu", description: "Japanese-specialized reasoning model based on Kimi K2.6 and tuned for Japanese language, culture, and business workflows", family: "sakana-namazu", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-03", last_updated: "2026-08-03", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 65536 }, links: [{ label: "Official product page", url: "https://sakana.ai/namazu/", type: "announcement" }, { label: "Official model documentation", url: "https://console.sakana.ai/models?model=sakana-namazu", type: "docs" }], benchmarks: [{ name: "AIME26", score: 96.67, source: "https://console.sakana.ai/models?model=sakana-namazu" }, { name: "MMLU-Pro", score: 90.33, source: "https://console.sakana.ai/models?model=sakana-namazu" }, { name: "LiveCodeBench v6", score: 90.33, source: "https://console.sakana.ai/models?model=sakana-namazu" }, { name: "JFBench", score: 37.4, source: "https://console.sakana.ai/models?model=sakana-namazu" }, { name: "Translation", score: 52.2, source: "https://console.sakana.ai/models?model=sakana-namazu" }, { name: "FairPoliticsQA", score: 56.3, source: "https://console.sakana.ai/models?model=sakana-namazu" }] }, "sakana/fugu-ultra": { id: "sakana/fugu-ultra", name: "Fugu Ultra", description: "Quality-first multi-agent model for hard research, analysis, and competitions", family: "fugu", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, release_date: "2026-06-15", last_updated: "2026-06-15", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 1e6 }, links: [{ label: "Official model catalog", url: "https://raw.githubusercontent.com/SakanaAI/fugu/refs/heads/main/configs/files/fugu.json", type: "docs" }], benchmarks: [{ name: "SWE Bench Pro", score: 73.7, source: "https://console.sakana.ai/models" }, { name: "Terminal Bench 2.1", score: 82.1, source: "https://console.sakana.ai/models" }, { name: "LiveCodeBench", score: 93.2, source: "https://console.sakana.ai/models" }, { name: "LiveCodeBench Pro", score: 90.8, source: "https://console.sakana.ai/models" }, { name: "Humanity\u2019s Last Exam", score: 50, source: "https://console.sakana.ai/models" }, { name: "CharXiv Reasoning", score: 86.6, source: "https://console.sakana.ai/models" }, { name: "GPQA Diamond", score: 95.5, source: "https://console.sakana.ai/models" }, { name: "SciCode", score: 58.7, source: "https://console.sakana.ai/models" }, { name: "\u03C43 Banking", score: 20.6, source: "https://console.sakana.ai/models" }, { name: "Long Context Reasoning", score: 73.3, source: "https://console.sakana.ai/models" }, { name: "MRCRv2", score: 93.6, source: "https://console.sakana.ai/models" }, { name: "CTI-REALM", score: 69.4, source: "https://console.sakana.ai/models" }] }, "perplexity/sonar-deep-research": { id: "perplexity/sonar-deep-research", name: "Sonar Deep Research", description: "Sonar search model for autonomous research and citation-backed long-form reports", family: "sonar", attachment: false, reasoning: true, tool_call: false, temperature: false, knowledge: "2025-01", release_date: "2025-02-01", last_updated: "2025-09-01", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 32768 } }, "perplexity/sonar": { id: "perplexity/sonar", name: "Sonar", description: "Fast web-grounded Sonar for current answers, citations, and lightweight retrieval", family: "sonar", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2025-09-01", release_date: "2024-01-01", last_updated: "2025-09-01", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 4096 }, benchmarks: [{ name: "SciCode", score: 22.9, metric: "percent correct", source: "https://openrouter.ai/perplexity/sonar/benchmarks", date: "2026-03-11" }] }, "perplexity/sonar-reasoning-pro": { id: "perplexity/sonar-reasoning-pro", name: "Sonar Reasoning Pro", description: "Web-grounded Sonar for multi-step research questions that need cited reasoning", family: "sonar-reasoning", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2025-09-01", release_date: "2024-01-01", last_updated: "2025-09-01", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 4096 } }, "perplexity/sonar-pro": { id: "perplexity/sonar-pro", name: "Sonar Pro", description: "Deeper Sonar search model with broader retrieval and stronger synthesis", family: "sonar-pro", attachment: true, reasoning: false, tool_call: false, temperature: true, knowledge: "2025-09-01", release_date: "2024-01-01", last_updated: "2025-09-01", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 8192 }, benchmarks: [{ name: "SciCode", score: 22.6, metric: "percent correct", source: "https://openrouter.ai/perplexity/sonar-pro/benchmarks", date: "2026-03-11" }] }, "zhipuai/glm-4.6": { id: "zhipuai/glm-4.6", name: "GLM-4.6", description: "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", family: "glm", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-09-30", last_updated: "2025-09-30", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.6" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 29.5, metric: "index", source: "https://openrouter.ai/z-ai/glm-4.6/benchmarks", date: "2026-05-22" }, { name: "SciCode", score: 38.4, metric: "percent correct", source: "https://openrouter.ai/z-ai/glm-4.6/benchmarks", date: "2026-05-22" }, { name: "Terminal-Bench Hard", score: 25, metric: "success rate", source: "https://openrouter.ai/z-ai/glm-4.6/benchmarks", date: "2026-05-22" }, { name: "SWE-Bench Pro", score: 9.67, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "zhipuai/glm-4.5-flash": { id: "zhipuai/glm-4.5-flash", name: "GLM-4.5-Flash", description: "Efficient GLM model for fast reasoning, coding, and agent workflows", family: "glm-flash", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-07-28", last_updated: "2025-07-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 98304 } }, "zhipuai/glm-5v-turbo": { id: "zhipuai/glm-5v-turbo", name: "GLM-5V-Turbo", description: "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", family: "glm", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-04-01", last_updated: "2026-04-01", modalities: { input: ["text", "image", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 131072 } }, "zhipuai/glm-4.7": { id: "zhipuai/glm-4.7", name: "GLM-4.7", description: "Mature GLM model for dependable coding, reasoning, and structured agent tasks", family: "glm", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-12-22", last_updated: "2025-12-22", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.7" }], benchmarks: [{ name: "SWE-Bench Verified", score: 73.8, metric: "resolved", source: "https://huggingface.co/zai-org/GLM-4.7" }, { name: "Terminal Bench 2.0", score: 33.4, metric: "score", source: "https://huggingface.co/zai-org/GLM-4.7" }] }, "zhipuai/glm-4.5v": { id: "zhipuai/glm-4.5v", name: "GLM-4.5V", description: "GLM vision model for visual reasoning, documents, and multimodal agents", family: "glm", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-08-11", last_updated: "2025-08-11", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 64e3, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.5V" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 10.9, metric: "index", source: "https://openrouter.ai/z-ai/glm-4.5v/benchmarks", date: "2026-04-29" }, { name: "SciCode", score: 22.1, metric: "percent correct", source: "https://openrouter.ai/z-ai/glm-4.5v/benchmarks", date: "2026-04-29" }, { name: "Terminal-Bench Hard", score: 5.3, metric: "success rate", source: "https://openrouter.ai/z-ai/glm-4.5v/benchmarks", date: "2026-04-29" }] }, "zhipuai/glm-4.6v-flash": { id: "zhipuai/glm-4.6v-flash", name: "GLM-4.6V-Flash", description: "Lightweight GLM vision model for visual reasoning, documents, and multimodal agents", family: "glm", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2025-12-08", last_updated: "2025-12-08", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.6V-Flash" }] }, "zhipuai/glm-4.7-flashx": { id: "zhipuai/glm-4.7-flashx", name: "GLM-4.7-FlashX", description: "Efficient GLM model for fast reasoning, coding, and agent workflows", family: "glm-flash", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-01-19", last_updated: "2026-01-19", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 2e5, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.7-Flash" }] }, "zhipuai/glm-5": { id: "zhipuai/glm-5", name: "GLM-5", description: "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", family: "glm", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-12", last_updated: "2026-02-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-5" }], benchmarks: [{ name: "SWE-Bench Verified", score: 72.8, metric: "resolved", source: "https://www.swebench.com/" }, { name: "SWE-Atlas Codebase QnA", score: 20.5, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 24.24, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 28.74, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }] }, "zhipuai/glm-5-turbo": { id: "zhipuai/glm-5-turbo", name: "GLM-5-Turbo", description: "Faster GLM-5 lane for coding agents that need lower latency", family: "glm", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-03-16", last_updated: "2026-03-16", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 131072 } }, "zhipuai/glm-5.3": { id: "zhipuai/glm-5.3", name: "GLM-5.3", description: "Flagship GLM model for long-horizon coding, agents, and complex project delivery", family: "glm", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-14", last_updated: "2026-08-14", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 131072 } }, "zhipuai/glm-5.2": { id: "zhipuai/glm-5.2", name: "GLM-5.2", description: "Open flagship GLM for long-horizon coding agents and million-token context work", family: "glm", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-06-13", last_updated: "2026-06-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-5.2" }], benchmarks: [{ name: "SWE-Bench Pro", score: 62.1, metric: "resolve rate", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Terminal-Bench", score: 82.7, metric: "success rate", harness: "Claude Code", version: "2.1", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "FrontierSWE", score: 74.4, metric: "dominance", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Humanity's Last Exam", score: 40.5, metric: "accuracy", dataset: "text-only subset", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Humanity's Last Exam", score: 54.7, metric: "accuracy", variant: "with tools", dataset: "text-only subset", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "CritPt", score: 20.9, metric: "accuracy", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "AIME", score: 99.2, metric: "accuracy", version: "2026", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "HMMT", score: 94.4, metric: "accuracy", version: "November 2025", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "HMMT", score: 92.5, metric: "accuracy", version: "February 2026", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "IMOAnswerBench", score: 91, metric: "accuracy", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "GPQA Diamond", score: 91.2, metric: "accuracy", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "NL2Repo", score: 48.9, metric: "resolve rate", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "DeepSWE", score: 46.2, metric: "resolve rate", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Program Bench", score: 63.7, metric: "score", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Terminal-Bench", score: 81, metric: "success rate", harness: "Terminus 2", version: "2.1", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "PostTrainBench", score: 34.3, metric: "score", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "SWE Marathon", score: 13, metric: "resolve rate", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "MCP Atlas", score: 76.8, metric: "score", dataset: "public subset", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Tool-Decathlon", score: 48.2, metric: "score", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }] }, "zhipuai/glm-4.6v": { id: "zhipuai/glm-4.6v", name: "GLM-4.6V", description: "GLM vision model for visual reasoning, documents, and multimodal agents", family: "glm", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-12-08", last_updated: "2025-12-08", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.6V" }] }, "zhipuai/glm-4.5-air": { id: "zhipuai/glm-4.5-air", name: "GLM-4.5-Air", description: "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", family: "glm-air", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-07-28", last_updated: "2025-07-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 98304 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.5-Air" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 23.8, metric: "index", source: "https://openrouter.ai/z-ai/glm-4.5-air/benchmarks", date: "2026-05-30" }, { name: "SciCode", score: 30.6, metric: "percent correct", source: "https://openrouter.ai/z-ai/glm-4.5-air/benchmarks", date: "2026-05-30" }, { name: "Terminal-Bench Hard", score: 20.5, metric: "success rate", source: "https://openrouter.ai/z-ai/glm-4.5-air/benchmarks", date: "2026-05-30" }] }, "zhipuai/glm-5.1": { id: "zhipuai/glm-5.1", name: "GLM-5.1", description: "Strong GLM coding model for agentic engineering, terminals, and repository generation", family: "glm", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-07", last_updated: "2026-04-07", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 2e5, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-5.1" }], benchmarks: [{ name: "Artificial Analysis Coding Agent Index", score: 52.7, metric: "average pass@1", harness: "Claude Code", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 73.2, metric: "pass@1", harness: "Claude Code", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 19.8, metric: "pass@1", harness: "Claude Code", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 65.1, metric: "pass@1", harness: "Claude Code", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }] }, "zhipuai/glm-4.7-flash": { id: "zhipuai/glm-4.7-flash", name: "GLM-4.7-Flash", description: "Budget GLM lane for fast coding help, routing, and everyday automation", family: "glm-flash", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-01-19", last_updated: "2026-01-19", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 2e5, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.7-Flash" }], benchmarks: [{ name: "SWE-Bench Verified", score: 59.2, metric: "resolved", source: "https://huggingface.co/zai-org/GLM-4.7-Flash" }] }, "zhipuai/glm-4.5": { id: "zhipuai/glm-4.5", name: "GLM-4.5", description: "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", family: "glm", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-07-28", last_updated: "2025-07-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 98304 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.5" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 26.3, metric: "index", source: "https://openrouter.ai/z-ai/glm-4.5/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 34.8, metric: "percent correct", source: "https://openrouter.ai/z-ai/glm-4.5/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 22, metric: "success rate", source: "https://openrouter.ai/z-ai/glm-4.5/benchmarks", date: "2026-03-11" }] }, "ibm/granite-4-h-micro": { id: "ibm/granite-4-h-micro", name: "Granite-4.0-H-Micro", description: "Compact open-weight hybrid Granite model for lightweight enterprise chat and tool calling", family: "granite", attachment: false, reasoning: false, tool_call: true, structured_output: true, temperature: true, release_date: "2025-10-02", last_updated: "2025-10-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/ibm-granite/granite-4.0-h-micro" }] }, "ibm/granite-4-h-small": { id: "ibm/granite-4-h-small", name: "Granite-4.0-H-Small", description: "Open-weight hybrid model for enterprise chat, coding, retrieval-augmented generation, and tool-calling workloads", family: "granite", attachment: false, reasoning: false, tool_call: true, structured_output: true, temperature: true, release_date: "2025-10-02", last_updated: "2025-10-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/ibm-granite/granite-4.0-h-small" }] }, "thinkingmachines/inkling-small": { id: "thinkingmachines/inkling-small", name: "Inkling Small", description: "Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio", family: "ling", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-07-30", last_updated: "2026-07-30", modalities: { input: ["text", "image", "audio"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 1048576 }, license: "Apache-2.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/thinkingmachines/Inkling-Small" }] }, "thinkingmachines/inkling": { id: "thinkingmachines/inkling", name: "Inkling", description: "Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio", family: "ling", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-07-15", last_updated: "2026-07-15", modalities: { input: ["text", "image", "audio"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 1048576 }, license: "Apache-2.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/thinkingmachines/Inkling" }] }, "mistral/magistral-small-2506": { id: "mistral/magistral-small-2506", name: "Magistral Small", description: "Open Mistral reasoning model for transparent step-by-step problem solving", family: "magistral", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-06-10", last_updated: "2025-06-10", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "Apache 2.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Magistral-Small-2506" }] }, "mistral/mistral-small-latest": { id: "mistral/mistral-small-latest", name: "Mistral Small (latest)", description: "Efficient Mistral model for fast chat, extraction, and production assistants", family: "mistral-small", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-06", release_date: "2026-03-16", last_updated: "2026-03-16", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 256e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Small-4-119B-2603" }] }, "mistral/devstral-medium-2507": { id: "mistral/devstral-medium-2507", name: "Devstral Medium", description: "Mistral coding agent model for repository tasks and software engineering workflows", family: "devstral", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-05", release_date: "2025-07-10", last_updated: "2025-07-10", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Verified", score: 61.6, metric: "resolved", source: "https://mistral.ai/news/devstral-2507", date: "2025-07-10" }] }, "mistral/mistral-small-3-1-24b-instruct-2503": { id: "mistral/mistral-small-3-1-24b-instruct-2503", name: "Mistral Small 3.1 24B", description: "Efficient multimodal model for instruction following, coding, reasoning, and function calling", family: "mistral-small", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-06", release_date: "2025-03-17", last_updated: "2025-03-17", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Small-3.1-24B-Instruct-2503" }] }, "mistral/mistral-large-2411": { id: "mistral/mistral-large-2411", name: "Mistral Large 2.1", description: "Flagship Mistral model for advanced reasoning, coding, and multilingual work", family: "mistral-large", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-11", release_date: "2024-11-18", last_updated: "2024-11-18", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Large-Instruct-2411" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 13.8, metric: "index", source: "https://openrouter.ai/mistralai/mistral-large-2407/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 29.2, metric: "percent correct", source: "https://openrouter.ai/mistralai/mistral-large-2407/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 6.1, metric: "success rate", source: "https://openrouter.ai/mistralai/mistral-large-2407/benchmarks", date: "2026-03-11" }] }, "mistral/mistral-nemo": { id: "mistral/mistral-nemo", name: "Mistral Nemo", description: "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment", family: "mistral-nemo", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-07", release_date: "2024-07-01", last_updated: "2024-07-01", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Nemo-Instruct-2407" }] }, "mistral/mistral-large-latest": { id: "mistral/mistral-large-latest", name: "Mistral Large (latest)", description: "Flagship Mistral model for advanced reasoning, coding, and multilingual work", family: "mistral-large", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-11", release_date: "2024-11-01", last_updated: "2025-12-02", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512" }] }, "mistral/ministral-8b-instruct-2410": { id: "mistral/ministral-8b-instruct-2410", name: "Ministral 8B Instruct", description: "Efficient open Mistral edge model for on-device chat and function calling", family: "ministral", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2024-10-16", last_updated: "2024-10-16", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "Mistral Research License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Ministral-8B-Instruct-2410" }] }, "mistral/devstral-small-2507": { id: "mistral/devstral-small-2507", name: "Devstral Small", description: "Mistral coding agent model for repository tasks and software engineering workflows", family: "devstral", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-05", release_date: "2025-07-10", last_updated: "2025-07-10", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Devstral-Small-2507" }], benchmarks: [{ name: "SWE-Bench Verified", score: 53.6, metric: "resolved", source: "https://mistral.ai/news/devstral-2507", date: "2025-07-10" }] }, "mistral/devstral-2512": { id: "mistral/devstral-2512", name: "Devstral 2", description: "Mistral's coding-agent model for repository work, terminal tasks, and software fixes", family: "devstral", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-12", release_date: "2025-12-09", last_updated: "2025-12-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Devstral-2-123B-Instruct-2512" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 23.7, metric: "index", source: "https://openrouter.ai/mistralai/devstral-2512/benchmarks", date: "2026-05-31" }, { name: "SciCode", score: 33.1, metric: "percent correct", source: "https://openrouter.ai/mistralai/devstral-2512/benchmarks", date: "2026-05-31" }, { name: "Terminal-Bench Hard", score: 18.9, metric: "success rate", source: "https://openrouter.ai/mistralai/devstral-2512/benchmarks", date: "2026-05-31" }] }, "mistral/mistral-small-2603": { id: "mistral/mistral-small-2603", name: "Mistral Small 4", description: "Fast Mistral production model for chat, extraction, and cost-sensitive agents", family: "mistral-small", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-06", release_date: "2026-03-16", last_updated: "2026-03-16", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 256e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Small-4-119B-2603" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 24.3, metric: "index", source: "https://openrouter.ai/mistralai/mistral-small-2603/benchmarks", date: "2026-06-01" }, { name: "SciCode", score: 38, metric: "percent correct", source: "https://openrouter.ai/mistralai/mistral-small-2603/benchmarks", date: "2026-06-01" }, { name: "Terminal-Bench Hard", score: 17.4, metric: "success rate", source: "https://openrouter.ai/mistralai/mistral-small-2603/benchmarks", date: "2026-06-01" }] }, "mistral/mistral-small-2506": { id: "mistral/mistral-small-2506", name: "Mistral Small 3.2", description: "Efficient Mistral model for fast chat, extraction, and production assistants", family: "mistral-small", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-03", release_date: "2025-06-20", last_updated: "2025-06-20", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Small-3.2-24B-Instruct-2506" }] }, "mistral/pixtral-large-latest": { id: "mistral/pixtral-large-latest", name: "Pixtral Large (latest)", description: "Mistral's larger vision model for document-heavy image understanding and chat", family: "pixtral", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-11", release_date: "2024-11-01", last_updated: "2024-11-04", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Pixtral-Large-Instruct-2411" }] }, "mistral/mistral-large-2512": { id: "mistral/mistral-large-2512", name: "Mistral Large 3", description: "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning", family: "mistral-large", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-11", release_date: "2024-11-01", last_updated: "2025-12-02", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 22.7, metric: "index", source: "https://openrouter.ai/mistralai/mistral-large-2512/benchmarks", date: "2026-06-04" }, { name: "SciCode", score: 36.2, metric: "percent correct", source: "https://openrouter.ai/mistralai/mistral-large-2512/benchmarks", date: "2026-06-04" }, { name: "Terminal-Bench Hard", score: 15.9, metric: "success rate", source: "https://openrouter.ai/mistralai/mistral-large-2512/benchmarks", date: "2026-06-04" }] }, "mistral/mistral-medium-2604": { id: "mistral/mistral-medium-2604", name: "Mistral Medium 3.5", description: "Balanced Mistral model for enterprise assistants, multilingual work, and tools", family: "mistral-medium", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-29", last_updated: "2026-04-29", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 77.6, metric: "resolved", source: "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B" }, { name: "\u03C4\xB3-Telecom", score: 91.4, metric: "accuracy", variant: "public preview", source: "https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5/", date: "2026-05-22" }] }, "mistral/mistral-medium-2505": { id: "mistral/mistral-medium-2505", name: "Mistral Medium 3", description: "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", family: "mistral-medium", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-05", release_date: "2025-05-07", last_updated: "2025-05-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 131072 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 13.6, metric: "index", source: "https://openrouter.ai/mistralai/mistral-medium-3/benchmarks", date: "2026-05-30" }, { name: "SciCode", score: 33.1, metric: "percent correct", source: "https://openrouter.ai/mistralai/mistral-medium-3/benchmarks", date: "2026-05-30" }, { name: "Terminal-Bench Hard", score: 3.8, metric: "success rate", source: "https://openrouter.ai/mistralai/mistral-medium-3/benchmarks", date: "2026-05-30" }] }, "mistral/devstral-medium-latest": { id: "mistral/devstral-medium-latest", name: "Devstral 2 (latest)", description: "Mistral coding agent model for repository tasks and software engineering workflows", family: "devstral", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-12", release_date: "2025-12-02", last_updated: "2025-12-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Devstral-2-123B-Instruct-2512" }] }, "mistral/pixtral-12b": { id: "mistral/pixtral-12b", name: "Pixtral 12B", description: "Mistral vision-language model for image understanding and multimodal chat", family: "pixtral", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-09", release_date: "2024-09-01", last_updated: "2024-09-01", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Pixtral-12B-2409" }] }, "mistral/voxtral-small-latest": { id: "mistral/voxtral-small-latest", name: "Voxtral Small (latest)", description: "Instruct model with native audio input for speech understanding and tool use", family: "voxtral", attachment: true, reasoning: false, tool_call: true, temperature: true, release_date: "2025-07-15", last_updated: "2025-07-15", modalities: { input: ["text", "audio"], output: ["text"] }, open_weights: true, limit: { context: 32e3, output: 32e3 } }, "mistral/codestral-22b-v0.1": { id: "mistral/codestral-22b-v0.1", name: "Codestral-22B-v0.1", description: "Open Mistral code model for fill-in-the-middle and 80+ programming languages", family: "codestral", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2024-05-29", last_updated: "2024-05-29", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 32768, output: 8192 }, license: "Mistral AI Non-Production License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Codestral-22B-v0.1" }] }, "mistral/codestral-latest": { id: "mistral/codestral-latest", name: "Codestral (latest)", description: "Mistral code model for completions, refactors, and developer IDE workflows", family: "codestral", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-10", release_date: "2024-05-29", last_updated: "2025-01-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Codestral-22B-v0.1" }], benchmarks: [{ name: "Aider Polyglot", score: 11.1, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-01-13" }] }, "mistral/magistral-medium-latest": { id: "mistral/magistral-medium-latest", name: "Magistral Medium (latest)", description: "Mistral reasoning model for transparent analysis, math, and complex decisions", family: "magistral-medium", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-06", release_date: "2025-03-17", last_updated: "2025-03-20", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 } }, "mistral/mistral-medium-latest": { id: "mistral/mistral-medium-latest", name: "Mistral Medium (latest)", description: "Balanced Mistral model for enterprise assistants, multilingual work, and tools", family: "mistral-medium", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-29", last_updated: "2026-04-29", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 77.6, metric: "resolved", source: "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B" }] }, "upstage/solar-pro2": { id: "upstage/solar-pro2", name: "Solar Pro 2", description: "Flagship model for demanding analysis, coding, and production agent workflows", family: "solar-pro", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03", release_date: "2025-05-20", last_updated: "2025-05-20", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 65536, output: 8192 } }, "upstage/solar-pro4": { id: "upstage/solar-pro4", name: "Solar Pro 4", description: "Upstage's flagship model, specialized for agentic use", family: "solar-pro", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-02", release_date: "2026-08-06", last_updated: "2026-08-06", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 524288, output: 131072 } }, "upstage/solar-pro3": { id: "upstage/solar-pro3", name: "Solar Pro 3", description: "Flagship model for demanding analysis, coding, and production agent workflows", family: "solar-pro", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03", release_date: "2026-01", last_updated: "2026-01", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 8192 } }, "minimax/MiniMax-M2.7": { id: "minimax/MiniMax-M2.7", name: "MiniMax-M2.7", description: "Open MiniMax flagship for coding agents, office automation, and complex environments", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-03-18", last_updated: "2026-03-18", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2.7" }], benchmarks: [{ name: "SWE-Bench Verified", score: 79.9, metric: "resolved", harness: "Claude Code", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "SWE-Bench Pro", score: 56.2, metric: "resolve rate", harness: "Claude Code", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "Terminal-Bench", score: 51.1, metric: "success rate", version: "2.1", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }] }, "minimax/MiniMax-M2.7-highspeed": { id: "minimax/MiniMax-M2.7-highspeed", name: "MiniMax-M2.7-highspeed", description: "Low-latency M2.7 variant for interactive coding plans and agent loops", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-03-18", last_updated: "2026-03-18", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2.7" }] }, "minimax/MiniMax-M2.1": { id: "minimax/MiniMax-M2.1", name: "MiniMax-M2.1", description: "Earlier MiniMax agent model for practical coding and productivity tasks", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-12-23", last_updated: "2025-12-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2.1" }], benchmarks: [{ name: "SWE-Bench Verified", score: 74, metric: "resolved", source: "https://huggingface.co/MiniMaxAI/MiniMax-M2.1" }, { name: "SWE-Bench Pro", score: 36.81, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "minimax/MiniMax-M2": { id: "minimax/MiniMax-M2", name: "MiniMax-M2", description: "Efficient open MiniMax model built for coding agents and tool-heavy workflows", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-10-27", last_updated: "2025-10-27", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 196608, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2" }], benchmarks: [{ name: "SWE-Bench Verified", score: 69.4, metric: "resolved", source: "https://huggingface.co/MiniMaxAI/MiniMax-M2" }] }, "minimax/MiniMax-M2.5-highspeed": { id: "minimax/MiniMax-M2.5-highspeed", name: "MiniMax-M2.5-highspeed", description: "High-speed MiniMax model for low-latency coding and agent workflows", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-13", last_updated: "2026-02-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2.5" }] }, "minimax/MiniMax-M3": { id: "minimax/MiniMax-M3", name: "MiniMax-M3", description: "MiniMax multimodal model for long-context coding, perception, and agent planning", family: "minimax", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-01", last_updated: "2026-06-01", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 512e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M3" }], benchmarks: [{ name: "SWE-Bench Verified", score: 80.5, metric: "resolved", harness: "Claude Code", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "SWE-Bench Pro", score: 59, metric: "resolve rate", harness: "Claude Code", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "Terminal-Bench", score: 66, metric: "success rate", version: "2.1", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "BrowseComp", score: 83.52, metric: "accuracy", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "MCP Atlas", score: 74.2, metric: "success rate", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "OSWorld-Verified", score: 70.06, metric: "success rate", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }] }, "minimax/MiniMax-M2-Her": { id: "minimax/MiniMax-M2-Her", name: "MiniMax-M2 Her", description: "MiniMax M2 variant tuned for conversational and character-driven agent interactions", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-01-23", last_updated: "2026-01-23", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 131e3 } }, "minimax/MiniMax-M2.5": { id: "minimax/MiniMax-M2.5", name: "MiniMax-M2.5", description: "Prior MiniMax coding model for agent workflows, office edits, and automation", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-12", last_updated: "2026-02-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2.5" }], benchmarks: [{ name: "SWE-Bench Verified", score: 75.8, metric: "resolved", source: "https://www.swebench.com/" }, { name: "SWE-Atlas Codebase QnA", score: 10.3, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 19.52, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 18.6, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }] } } };
48228
48351
 
@@ -48283,7 +48406,7 @@ function diskCacheFresh() {
48283
48406
  }
48284
48407
  }
48285
48408
  async function fetchFresh() {
48286
- const dispatcher = proxyDispatcher(process.env.https_proxy || process.env.HTTPS_PROXY || process.env.http_proxy || process.env.HTTP_PROXY);
48409
+ const dispatcher = proxyDispatcher(process.env.https_proxy || process.env.HTTPS_PROXY || process.env.http_proxy || process.env.HTTP_PROXY, 15e3);
48287
48410
  const attempts = dispatcher ? [{ opts: { dispatcher }, label: "via proxy" }, { opts: {}, label: "direct" }] : [{ opts: {}, label: "direct" }];
48288
48411
  for (const { opts, label } of attempts) {
48289
48412
  try {
@@ -49950,8 +50073,17 @@ var SessionStore = class {
49950
50073
  async loadAll() {
49951
50074
  const out = /* @__PURE__ */ new Map();
49952
50075
  if (!this.enabled) return out;
50076
+ let clamped = 0;
49953
50077
  for (const [id, envelope] of await this.store.loadAll()) {
49954
- out.set(id, buildSession(envelope.payload));
50078
+ const session = buildSession(envelope.payload);
50079
+ if (hasNegativePersistedTokens(envelope.payload)) {
50080
+ await this.store.writeNow(id, () => buildRecord(session));
50081
+ clamped++;
50082
+ }
50083
+ out.set(id, session);
50084
+ }
50085
+ if (clamped > 0) {
50086
+ log("info", `[persist] one-time migration (#408): clamped negative lastInputTokens/contextTokens in ${clamped} session(s) to 0`);
49955
50087
  }
49956
50088
  return out;
49957
50089
  }
@@ -50073,7 +50205,14 @@ var SessionStore = class {
50073
50205
  meta?.protocol ? this.store.loadSync(id, relPathFor(id)) : null
50074
50206
  ];
50075
50207
  for (const envelope of envelopes) {
50076
- if (envelope) return buildSession(envelope.payload);
50208
+ if (envelope) {
50209
+ const session = buildSession(envelope.payload);
50210
+ if (hasNegativePersistedTokens(envelope.payload)) {
50211
+ this.scheduleSave(session);
50212
+ log("info", `[persist] clamped negative token stats on reload for ${id} (#408)`);
50213
+ }
50214
+ return session;
50215
+ }
50077
50216
  }
50078
50217
  return null;
50079
50218
  }
@@ -50115,13 +50254,15 @@ var SessionStore = class {
50115
50254
  }
50116
50255
  };
50117
50256
  function buildRecord(session) {
50257
+ const snapshot = boundedFoldedSnapshot(session);
50118
50258
  return {
50119
50259
  version: PERSIST_VERSION,
50120
50260
  savedAt: Date.now(),
50121
50261
  id: session.id,
50122
50262
  meta: { ...session.meta },
50123
50263
  stats: { ...session.stats },
50124
- messages: session.lastMessages,
50264
+ messages: snapshot,
50265
+ messagesFolded: snapshot ? true : void 0,
50125
50266
  metadata: { ...session.metadata },
50126
50267
  state: session.state,
50127
50268
  blockContents: Object.fromEntries(session.blockContents),
@@ -50161,10 +50302,14 @@ function buildSession(parsed) {
50161
50302
  cachedTokens: stats.cachedTokens ?? parsed.cachedTokens ?? 0,
50162
50303
  outputTokens: stats.outputTokens ?? parsed.outputTokens ?? 0,
50163
50304
  cacheSamples: stats.cacheSamples ?? parsed.cacheSamples ?? 0,
50164
- lastInputTokens: stats.lastInputTokens ?? parsed.lastInputTokens ?? 0,
50305
+ // #408: clamp at restore — pre-clamp versions persisted negative
50306
+ // values (lastInputTokens = total − credit before the Math.max
50307
+ // guard existed) which would otherwise revive after upgrade and
50308
+ // feed the /acp panel + web stats as negative percentages.
50309
+ lastInputTokens: Math.max(0, stats.lastInputTokens ?? parsed.lastInputTokens ?? 0),
50165
50310
  // In-memory only — a fresh process has no pending compress fold.
50166
50311
  compressCreditTokens: 0,
50167
- contextTokens: stats.contextTokens ?? parsed.contextTokens ?? 0
50312
+ contextTokens: Math.max(0, stats.contextTokens ?? parsed.contextTokens ?? 0)
50168
50313
  },
50169
50314
  metadata: parsed.metadata ?? {},
50170
50315
  state: mergeState(parsed.state),
@@ -50178,6 +50323,7 @@ function buildSession(parsed) {
50178
50323
  restored: true,
50179
50324
  blockContents,
50180
50325
  lastMessages: Array.isArray(parsed.messages) ? parsed.messages : void 0,
50326
+ lastMessagesFolded: parsed.messagesFolded === true,
50181
50327
  inFlight: 0,
50182
50328
  persisted: true
50183
50329
  };
@@ -50187,6 +50333,12 @@ function isValidRecord(parsed) {
50187
50333
  const r = parsed;
50188
50334
  return typeof r.id === "string" && typeof r.state === "object" && r.state !== null && Array.isArray(r.state.blocks);
50189
50335
  }
50336
+ function hasNegativePersistedTokens(parsed) {
50337
+ const stats = parsed.stats ?? {};
50338
+ const last = stats.lastInputTokens ?? parsed.lastInputTokens;
50339
+ const ctx = stats.contextTokens ?? parsed.contextTokens;
50340
+ return typeof last === "number" && last < 0 || typeof ctx === "number" && ctx < 0;
50341
+ }
50190
50342
  function defaultDir() {
50191
50343
  return sessionsDir();
50192
50344
  }
@@ -50203,6 +50355,36 @@ function persistEnabled() {
50203
50355
  if (env === "0" || env === "false") return false;
50204
50356
  return true;
50205
50357
  }
50358
+ function persistTailTokens() {
50359
+ const env = process.env.BILI_PERSIST_TAIL_TOKENS;
50360
+ if (env) {
50361
+ const n = Number.parseInt(env, 10);
50362
+ if (Number.isFinite(n) && n >= 0) return n;
50363
+ }
50364
+ return 16384;
50365
+ }
50366
+ function boundedFoldedSnapshot(session) {
50367
+ const msgs = session.lastMessages;
50368
+ if (!msgs || msgs.length === 0) return void 0;
50369
+ const budget = persistTailTokens();
50370
+ if (budget === 0) return void 0;
50371
+ let view = prune(msgs, session.state);
50372
+ let total = 0;
50373
+ for (const m2 of view) total += defaultCountTokens(m2.text ?? "");
50374
+ if (total > budget) {
50375
+ let acc = 0;
50376
+ let start = 0;
50377
+ for (let i = view.length - 1; i >= 0; i--) {
50378
+ acc += defaultCountTokens(view[i].text ?? "");
50379
+ if (acc > budget) {
50380
+ start = Math.min(i + 1, view.length - 1);
50381
+ break;
50382
+ }
50383
+ }
50384
+ if (start > 0) view = view.slice(start);
50385
+ }
50386
+ return view;
50387
+ }
50206
50388
  function epermAlertThreshold() {
50207
50389
  const env = process.env.BILI_PERSIST_EPERM_ALERT_THRESHOLD;
50208
50390
  if (env) {
@@ -50322,7 +50504,10 @@ function peekSession(id) {
50322
50504
  return sessions.get(id);
50323
50505
  }
50324
50506
  function snapshotMessages(session, messages) {
50325
- if (messages.length > 0) session.lastMessages = messages;
50507
+ if (messages.length > 0) {
50508
+ session.lastMessages = messages;
50509
+ session.lastMessagesFolded = false;
50510
+ }
50326
50511
  }
50327
50512
  function markDirty(session) {
50328
50513
  getStore().scheduleSave(session);
@@ -50335,6 +50520,8 @@ function resetSessionCompression(session) {
50335
50520
  session.blockContents.clear();
50336
50521
  session.stats.lastInputTokens = 0;
50337
50522
  session.stats.contextTokens = 0;
50523
+ session.hostCreditTokens = 0;
50524
+ session.hostContextTokens = 0;
50338
50525
  session.metadata.nativeCompactionAt = Date.now();
50339
50526
  markDirty(session);
50340
50527
  }
@@ -50448,6 +50635,50 @@ function withStagedCompressGuidance(text) {
50448
50635
  return text + STAGED_COMPRESS_GUIDANCE;
50449
50636
  }
50450
50637
 
50638
+ // src/acp-status.ts
50639
+ function handleAcpStatus(args, ctx) {
50640
+ const scope = typeof args.scope === "string" ? args.scope : void 0;
50641
+ const view = typeof args.view === "string" ? args.view : void 0;
50642
+ const tool = typeof args.tool === "string" ? args.tool : void 0;
50643
+ const sort = typeof args.sort === "string" ? args.sort : void 0;
50644
+ const limit = typeof args.limit === "number" ? args.limit : void 0;
50645
+ const base = buildStatusReport(ctx.session.state, ctx.messages, defaultCountTokens, { scope, view, tool, sort, limit });
50646
+ if (scope) return base;
50647
+ const extra = [];
50648
+ try {
50649
+ const turn = ctx.core.processTurn({
50650
+ messages: ctx.messages,
50651
+ state: ctx.session.state,
50652
+ config: ctx.config,
50653
+ tokenCount: ctx.session.stats.lastInputTokens,
50654
+ renderTags: "none"
50655
+ });
50656
+ const nudge = turn.nudge;
50657
+ if (nudge) {
50658
+ extra.push("");
50659
+ extra.push(nudge.shouldInject ? `Nudge: ACTIVE \u2014 ${nudge.reason}` : `Nudge: idle \u2014 ${nudge.reason}`);
50660
+ const ranges = viableRanges(nudge.compressibleRanges);
50661
+ const protectedRanges = nudge.protectedRanges ?? [];
50662
+ if (ranges.length > 0 || protectedRanges.length > 0) {
50663
+ extra.push("");
50664
+ extra.push(formatRanges(ranges, protectedRanges));
50665
+ }
50666
+ }
50667
+ } catch {
50668
+ }
50669
+ const archive = preCompactionArchiveOf(ctx.session);
50670
+ const archivedIds = Object.keys(archive);
50671
+ if (archivedIds.length > 0) {
50672
+ extra.push("");
50673
+ extra.push(`PRE-COMPACTION ARCHIVE \u2014 ${archivedIds.length} block(s): content was replaced by the client's native compaction summary, so it is no longer in the session history and decompress is unavailable.`);
50674
+ for (const id of archivedIds) {
50675
+ extra.push(` ${id} \u2014 ${archive[id].reason}`);
50676
+ }
50677
+ }
50678
+ return extra.length > 0 ? `${base}
50679
+ ${extra.join("\n")}` : base;
50680
+ }
50681
+
50451
50682
  // src/decompress-shared.ts
50452
50683
  import { mkdirSync as mkdirSync4, unlinkSync as unlinkSync2, writeFileSync as writeFileSync3 } from "fs";
50453
50684
  import { dirname as dirname2, join as join2 } from "path";
@@ -50769,7 +51000,7 @@ function executeAnthropicProxyTool(toolName, args, ctx) {
50769
51000
  ${lines.join("\n\n")}`;
50770
51001
  }
50771
51002
  if (toolName === "acp_status") {
50772
- return buildStatusReport(ctx.session.state, ctx.messages, defaultCountTokens);
51003
+ return handleAcpStatus(args, ctx);
50773
51004
  }
50774
51005
  return `[Unknown proxy tool: ${toolName}]`;
50775
51006
  }
@@ -50876,6 +51107,7 @@ var CHUNK_FRACTION = 0.6;
50876
51107
  var MIN_CHUNK_TOKENS = 2e3;
50877
51108
  var MIN_SUMMARY_CHARS = 50;
50878
51109
  var MAX_SUMMARY_OUTPUT_TOKENS = 8192;
51110
+ var MAX_SUMMARY_CALLS_PER_PREFLIGHT = 8;
50879
51111
  function refMaps(messages, state) {
50880
51112
  const refToIdx = /* @__PURE__ */ new Map();
50881
51113
  const idxToRef = /* @__PURE__ */ new Map();
@@ -50896,6 +51128,25 @@ function estimateCoreMessages(messages) {
50896
51128
  for (const m2 of messages) tokens += defaultCountTokens(m2.text ?? "");
50897
51129
  return tokens;
50898
51130
  }
51131
+ var NON_TEXT_BODY_KEYS = /* @__PURE__ */ new Set(["data", "url", "b64_json", "file_data"]);
51132
+ function estimateRawBodyTokens(parsed) {
51133
+ let tokens = 0;
51134
+ const walk = (value, key) => {
51135
+ if (typeof value === "string") {
51136
+ if (!key || !NON_TEXT_BODY_KEYS.has(key)) tokens += defaultCountTokens(value);
51137
+ return;
51138
+ }
51139
+ if (Array.isArray(value)) {
51140
+ for (const item of value) walk(item, key);
51141
+ return;
51142
+ }
51143
+ if (value && typeof value === "object") {
51144
+ for (const [k2, v2] of Object.entries(value)) walk(v2, k2);
51145
+ }
51146
+ };
51147
+ walk(parsed);
51148
+ return tokens;
51149
+ }
50899
51150
  function rangeChars2(messages, startIdx, endIdx) {
50900
51151
  let chars = 0;
50901
51152
  for (let i = startIdx; i <= endIdx && i < messages.length; i++) {
@@ -51003,15 +51254,28 @@ TASK: The conversation segment below (messages ${startRef}\u2013${endRef}) must
51003
51254
  }
51004
51255
  }
51005
51256
  var ABORTED_FAILURE = { kind: "aborted", detail: "the client disconnected during preflight compression" };
51257
+ function relaxedConfig(config) {
51258
+ return { ...config, preserveRecentMessages: 0, preserveRecentTokens: 0 };
51259
+ }
51260
+ function noEmergencyTruncate(config) {
51261
+ return { ...config, modelContextLimit: config.modelContextLimit * 100 };
51262
+ }
51006
51263
  async function preflightCompress(deps, messages) {
51007
51264
  const limit = deps.config.modelContextLimit;
51008
- const result = { compressedRanges: 0, savedTokens: 0, payloadEstimate: estimateCoreMessages(messages) + (deps.imageFloor ?? 0) };
51265
+ const result = { compressedRanges: 0, savedTokens: 0, payloadEstimate: estimateCoreMessages(messages) + (deps.imageFloor ?? 0) + (deps.wireOverhead ?? 0) };
51009
51266
  if (limit <= 0) return result;
51010
51267
  const budget = Math.max(MIN_CHUNK_TOKENS, Math.floor(limit * CHUNK_FRACTION));
51011
51268
  const minChars = deps.config.compress.minCompressRange;
51012
51269
  let currentTokens = deps.session.stats.lastInputTokens;
51013
51270
  let startTokens = -1;
51014
51271
  let failure;
51272
+ let activeConfig = deps.config;
51273
+ let relaxed = false;
51274
+ const relaxedExhaustedDetail = `the payload still exceeds the window after folding everything compressible, including the soft-protected recent zone (last ${deps.config.preserveRecentMessages} messages + most recent user message), which was relaxed under overflow; hard protectedTools remain excluded. Raise the model context window or restart the session to recover.`;
51275
+ const skipSet = /* @__PURE__ */ new Set();
51276
+ let summaryCalls = 0;
51277
+ let budgetHit = false;
51278
+ let rangesTried = 0;
51015
51279
  for (let round = 0; round < MAX_PREFLIGHT_ROUNDS; round++) {
51016
51280
  if (deps.signal?.aborted) {
51017
51281
  failure = ABORTED_FAILURE;
@@ -51020,91 +51284,127 @@ async function preflightCompress(deps, messages) {
51020
51284
  const turn = deps.core.processTurn({
51021
51285
  messages,
51022
51286
  state: deps.session.state,
51023
- config: deps.config,
51287
+ config: noEmergencyTruncate(activeConfig),
51024
51288
  tokenCount: currentTokens,
51025
51289
  renderTags: "text-only"
51026
51290
  });
51027
51291
  deps.session.state = turn.state;
51028
- currentTokens = Math.max(deps.session.stats.lastInputTokens, estimateCoreMessages(turn.messages) + (deps.imageFloor ?? 0));
51029
- result.payloadEstimate = estimateCoreMessages(turn.messages) + (deps.imageFloor ?? 0);
51292
+ currentTokens = Math.max(deps.session.stats.lastInputTokens, estimateCoreMessages(turn.messages) + (deps.imageFloor ?? 0) + (deps.wireOverhead ?? 0));
51293
+ result.payloadEstimate = estimateCoreMessages(turn.messages) + (deps.imageFloor ?? 0) + (deps.wireOverhead ?? 0);
51030
51294
  if (startTokens < 0) startTokens = currentTokens;
51031
51295
  if (currentTokens < limit) break;
51032
51296
  const ranges = viableRanges(turn.nudge?.compressibleRanges ?? []);
51033
51297
  if (ranges.length === 0) {
51034
- failure = { kind: "exhausted", detail: "no compressible ranges remain in the conversation" };
51035
- break;
51036
- }
51037
- const range = [...ranges].sort((a, b2) => refNum3(a.startRef) - refNum3(b2.startRef))[0];
51038
- const { refToIdx } = refMaps(messages, deps.session.state);
51039
- const startIdx = refToIdx.get(range.startRef);
51040
- const endIdx = refToIdx.get(range.endRef);
51041
- if (startIdx === void 0 || endIdx === void 0 || startIdx > endIdx) {
51042
- failure = { kind: "exhausted", detail: "the compressible range no longer resolves to payload messages" };
51298
+ if (!relaxed && result.payloadEstimate >= limit) {
51299
+ activeConfig = relaxedConfig(deps.config);
51300
+ relaxed = true;
51301
+ summaryCalls = 0;
51302
+ budgetHit = false;
51303
+ deps.log("warn", "[preflight] no compressible ranges outside the protected recent zone; relaxing soft protection (preserveRecentMessages/Tokens -> 0) and retrying");
51304
+ continue;
51305
+ }
51306
+ failure = { kind: "exhausted", detail: relaxed ? relaxedExhaustedDetail : "no compressible ranges remain in the conversation" };
51043
51307
  break;
51044
51308
  }
51309
+ const ordered = [...ranges].sort((a, b2) => refNum3(a.startRef) - refNum3(b2.startRef));
51045
51310
  let appliedThisRound = 0;
51046
- for (const [cs2, ce2] of splitChunks(messages, startIdx, endIdx, budget)) {
51311
+ for (const range of ordered) {
51047
51312
  if (currentTokens < limit) break;
51048
51313
  if (deps.signal?.aborted) {
51049
51314
  failure = ABORTED_FAILURE;
51050
51315
  break;
51051
51316
  }
51052
- const maps = refMaps(messages, deps.session.state);
51053
- const startRef = maps.idxToRef.get(cs2);
51054
- const endRef = maps.idxToRef.get(ce2);
51055
- if (!startRef || !endRef) continue;
51056
- if (rangeChars2(messages, cs2, ce2) < minChars) continue;
51057
- const content = renderRange(messages, cs2, ce2);
51058
- if (content.length === 0) continue;
51059
- let summary;
51060
- try {
51061
- summary = await summarizeRange(deps, content, startRef, endRef);
51062
- } catch (err2) {
51063
- if (err2 instanceof UpstreamHttpError) {
51064
- failure = {
51065
- kind: "upstream",
51066
- status: err2.status,
51067
- detail: err2.status === 429 ? `the summarization call was rate-limited by the upstream (HTTP 429)` : `the summarization call was rejected by the upstream (HTTP ${err2.status})`
51068
- };
51069
- deps.log("warn", `[preflight] summarization failed: HTTP ${err2.status} ${err2.body.slice(0, 200)}`);
51070
- } else if (deps.signal?.aborted) {
51317
+ if (budgetHit) break;
51318
+ const skipKey = `${range.startRef}:${range.endRef}`;
51319
+ if (skipSet.has(skipKey)) continue;
51320
+ const { refToIdx } = refMaps(messages, deps.session.state);
51321
+ const startIdx = refToIdx.get(range.startRef);
51322
+ const endIdx = refToIdx.get(range.endRef);
51323
+ if (startIdx === void 0 || endIdx === void 0 || startIdx > endIdx) {
51324
+ skipSet.add(skipKey);
51325
+ continue;
51326
+ }
51327
+ rangesTried += 1;
51328
+ for (const [cs2, ce2] of splitChunks(messages, startIdx, endIdx, budget)) {
51329
+ if (currentTokens < limit) break;
51330
+ if (deps.signal?.aborted) {
51071
51331
  failure = ABORTED_FAILURE;
51072
- deps.log("warn", `[preflight] summarization aborted: client disconnected`);
51073
- } else {
51074
- failure = { kind: "upstream", detail: `the summarization call failed: ${String(err2)}` };
51075
- deps.log("warn", `[preflight] summarization failed: ${String(err2)}`);
51332
+ break;
51076
51333
  }
51334
+ if (budgetHit) break;
51335
+ const maps = refMaps(messages, deps.session.state);
51336
+ const startRef = maps.idxToRef.get(cs2);
51337
+ const endRef = maps.idxToRef.get(ce2);
51338
+ if (!startRef || !endRef) continue;
51339
+ if (rangeChars2(messages, cs2, ce2) < minChars) continue;
51340
+ const content = renderRange(messages, cs2, ce2);
51341
+ if (content.length === 0) continue;
51342
+ if (summaryCalls >= MAX_SUMMARY_CALLS_PER_PREFLIGHT) {
51343
+ budgetHit = true;
51344
+ break;
51345
+ }
51346
+ summaryCalls += 1;
51347
+ let summary;
51348
+ try {
51349
+ summary = await summarizeRange(deps, content, startRef, endRef);
51350
+ } catch (err2) {
51351
+ if (err2 instanceof UpstreamHttpError) {
51352
+ failure = {
51353
+ kind: "upstream",
51354
+ status: err2.status,
51355
+ detail: err2.status === 429 ? `the summarization call was rate-limited by the upstream (HTTP 429)` : `the summarization call was rejected by the upstream (HTTP ${err2.status})`
51356
+ };
51357
+ deps.log("warn", `[preflight] summarization failed: HTTP ${err2.status} ${err2.body.slice(0, 200)}`);
51358
+ } else if (deps.signal?.aborted) {
51359
+ failure = ABORTED_FAILURE;
51360
+ deps.log("warn", `[preflight] summarization aborted: client disconnected`);
51361
+ } else {
51362
+ failure = { kind: "upstream", detail: `the summarization call failed: ${String(err2)}` };
51363
+ deps.log("warn", `[preflight] summarization failed: ${String(err2)}`);
51364
+ }
51365
+ break;
51366
+ }
51367
+ if (!summary) {
51368
+ deps.log("warn", `[preflight] range ${skipKey} produced no usable summary; skipping it`);
51369
+ skipSet.add(skipKey);
51370
+ break;
51371
+ }
51372
+ const ctx = {
51373
+ core: deps.core,
51374
+ config: activeConfig,
51375
+ messages,
51376
+ session: deps.session,
51377
+ log: (msg) => deps.log("info", msg)
51378
+ };
51379
+ const creditBefore = deps.session.stats.compressCreditTokens;
51380
+ const applied = applyRanges(parseCompressInput({ content: [{ startId: startRef, endId: endRef, summary, topic: "preflight overflow compress" }] }), ctx);
51381
+ if (applied.startsWith("[Compression FAILED")) {
51382
+ deps.log("warn", `[preflight] ${applied}`);
51383
+ skipSet.add(skipKey);
51384
+ break;
51385
+ }
51386
+ const compressed = deps.session.stats.compressCreditTokens - creditBefore;
51387
+ currentTokens = Math.max(0, currentTokens - compressed + defaultCountTokens(summary));
51388
+ deps.session.stats.lastInputTokens += defaultCountTokens(summary);
51389
+ appliedThisRound += 1;
51390
+ result.compressedRanges += 1;
51077
51391
  break;
51078
51392
  }
51079
- if (!summary) continue;
51080
- const ctx = {
51081
- core: deps.core,
51082
- config: deps.config,
51083
- messages,
51084
- session: deps.session,
51085
- log: (msg) => deps.log("info", msg)
51086
- };
51087
- const creditBefore = deps.session.stats.compressCreditTokens;
51088
- const applied = applyRanges(parseCompressInput({ content: [{ startId: startRef, endId: endRef, summary, topic: "preflight overflow compress" }] }), ctx);
51089
- if (applied.startsWith("[Compression FAILED")) {
51090
- deps.log("warn", `[preflight] ${applied}`);
51091
- continue;
51092
- }
51093
- const compressed = deps.session.stats.compressCreditTokens - creditBefore;
51094
- currentTokens = Math.max(0, currentTokens - compressed + defaultCountTokens(summary));
51095
- deps.session.stats.lastInputTokens += defaultCountTokens(summary);
51096
- appliedThisRound += 1;
51097
- result.compressedRanges += 1;
51098
- }
51099
- if (appliedThisRound === 0) {
51100
- if (!failure) {
51101
- failure = { kind: "exhausted", detail: "no range could be compressed (chunks below minCompressRange or the summarization responses were unusable)" };
51102
- }
51103
- break;
51393
+ if (appliedThisRound > 0) break;
51394
+ if (failure || budgetHit) break;
51104
51395
  }
51396
+ if (appliedThisRound === 0) break;
51105
51397
  }
51106
- if (!failure && currentTokens >= limit) {
51107
- failure = { kind: "exhausted", detail: `the compress budget was exhausted after ${MAX_PREFLIGHT_ROUNDS} rounds` };
51398
+ if (currentTokens >= limit && !failure) {
51399
+ if (budgetHit) {
51400
+ failure = { kind: "exhausted", detail: `the preflight summarization budget (${MAX_SUMMARY_CALLS_PER_PREFLIGHT} calls per protection regime) was exhausted before the payload fit the window` };
51401
+ } else if (relaxed && result.compressedRanges > 0) {
51402
+ failure = { kind: "exhausted", detail: relaxedExhaustedDetail };
51403
+ } else if (result.compressedRanges === 0) {
51404
+ failure = { kind: "exhausted", detail: `no range could be compressed across ${rangesTried} viable range${rangesTried === 1 ? "" : "s"} (each was below minCompressRange, had an unusable summary, or failed to apply)` };
51405
+ } else {
51406
+ failure = { kind: "exhausted", detail: `the compress budget was exhausted after ${MAX_PREFLIGHT_ROUNDS} rounds` };
51407
+ }
51108
51408
  }
51109
51409
  if (result.compressedRanges > 0) deps.session.stats.lastInputTokens = currentTokens;
51110
51410
  result.savedTokens = Math.max(0, startTokens - currentTokens);
@@ -51634,7 +51934,7 @@ function reapOrphanBlocks(session, visible, deactivate) {
51634
51934
  }
51635
51935
 
51636
51936
  // src/instance.ts
51637
- import { randomUUID as randomUUID3 } from "crypto";
51937
+ import { createHash as createHash4, randomUUID as randomUUID3 } from "crypto";
51638
51938
  import fs3 from "fs";
51639
51939
  import path7 from "path";
51640
51940
  function isProxyInstanceFile(v2) {
@@ -51680,14 +51980,13 @@ function readProxyInstanceFile(file) {
51680
51980
  function instanceFilePath() {
51681
51981
  return path7.join(stateDir(), "proxy-origin");
51682
51982
  }
51683
- function atomicWriteInstanceFile(info, file) {
51684
- const filePath = file ?? instanceFilePath();
51983
+ function atomicWriteJson(obj, filePath) {
51685
51984
  fs3.mkdirSync(path7.dirname(filePath), { recursive: true });
51686
51985
  const tempPath = `${filePath}.${process.pid}.${randomUUID3()}.tmp`;
51687
51986
  let descriptor;
51688
51987
  try {
51689
51988
  descriptor = fs3.openSync(tempPath, "wx", 420);
51690
- fs3.writeSync(descriptor, JSON.stringify(info) + "\n", null, "utf8");
51989
+ fs3.writeSync(descriptor, JSON.stringify(obj) + "\n", null, "utf8");
51691
51990
  fs3.fsyncSync(descriptor);
51692
51991
  fs3.closeSync(descriptor);
51693
51992
  descriptor = void 0;
@@ -51706,6 +52005,9 @@ function atomicWriteInstanceFile(info, file) {
51706
52005
  throw error;
51707
52006
  }
51708
52007
  }
52008
+ function atomicWriteInstanceFile(info, file) {
52009
+ atomicWriteJson(info, file ?? instanceFilePath());
52010
+ }
51709
52011
  function clearProxyInstanceFile(instanceId, file) {
51710
52012
  const filePath = file ?? instanceFilePath();
51711
52013
  const current = readProxyInstanceFile(filePath);
@@ -51725,44 +52027,99 @@ function isPidAlive(pid) {
51725
52027
  return err2.code === "EPERM";
51726
52028
  }
51727
52029
  }
51728
- function registryFilePath() {
52030
+ function registryDirPath() {
52031
+ return path7.join(stateDir(), "instances");
52032
+ }
52033
+ function legacyRegistryFilePath() {
51729
52034
  return path7.join(stateDir(), "instances.json");
51730
52035
  }
51731
- function registerInstanceAndWarn(entry, warn) {
51732
- const file = registryFilePath();
51733
- const entries = [];
52036
+ var SAFE_NAME_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*$/;
52037
+ function safeRegistryName(instanceId) {
52038
+ if (instanceId !== "." && instanceId !== ".." && instanceId.length <= 128 && SAFE_NAME_RE.test(instanceId)) {
52039
+ return instanceId;
52040
+ }
52041
+ return createHash4("sha256").update(instanceId).digest("hex");
52042
+ }
52043
+ function registryEntryFile(instanceId) {
52044
+ return path7.join(registryDirPath(), `${safeRegistryName(instanceId)}.json`);
52045
+ }
52046
+ function safeReadJson2(file) {
51734
52047
  try {
51735
- const parsed = JSON.parse(fs3.readFileSync(file, "utf8"));
51736
- if (Array.isArray(parsed.instances)) {
51737
- for (const e of parsed.instances) {
51738
- if (e && typeof e.instanceId === "string" && e.instanceId !== entry.instanceId && isPidAlive(e.pid)) {
51739
- entries.push(e);
51740
- }
52048
+ return JSON.parse(fs3.readFileSync(file, "utf8"));
52049
+ } catch {
52050
+ return void 0;
52051
+ }
52052
+ }
52053
+ function coerceEntry(value) {
52054
+ if (!value || typeof value !== "object") return void 0;
52055
+ const o = value;
52056
+ if (typeof o.instanceId !== "string" || o.instanceId === "") return void 0;
52057
+ return {
52058
+ instanceId: o.instanceId,
52059
+ pid: typeof o.pid === "number" ? o.pid : 0,
52060
+ port: typeof o.port === "number" ? o.port : 0,
52061
+ origin: typeof o.origin === "string" ? o.origin : "",
52062
+ startedAt: typeof o.startedAt === "number" ? o.startedAt : 0
52063
+ };
52064
+ }
52065
+ function readMarkerNames() {
52066
+ try {
52067
+ return fs3.readdirSync(registryDirPath());
52068
+ } catch {
52069
+ return [];
52070
+ }
52071
+ }
52072
+ function readAllRegistryEntries() {
52073
+ const seen = /* @__PURE__ */ new Set();
52074
+ const out = [];
52075
+ for (const name of readMarkerNames()) {
52076
+ if (!name.endsWith(".json")) continue;
52077
+ const entry = coerceEntry(safeReadJson2(path7.join(registryDirPath(), name)));
52078
+ if (entry && !seen.has(entry.instanceId)) {
52079
+ seen.add(entry.instanceId);
52080
+ out.push(entry);
52081
+ }
52082
+ }
52083
+ const legacy = safeReadJson2(legacyRegistryFilePath());
52084
+ if (legacy && Array.isArray(legacy.instances)) {
52085
+ for (const raw of legacy.instances) {
52086
+ const entry = coerceEntry(raw);
52087
+ if (entry && !seen.has(entry.instanceId)) {
52088
+ seen.add(entry.instanceId);
52089
+ out.push(entry);
51741
52090
  }
51742
52091
  }
51743
- } catch {
51744
52092
  }
51745
- for (const other of entries) {
52093
+ return out;
52094
+ }
52095
+ function reapDeadMarkers(ours) {
52096
+ for (const name of readMarkerNames()) {
52097
+ if (!name.endsWith(".json")) continue;
52098
+ const file = path7.join(registryDirPath(), name);
52099
+ const entry = coerceEntry(safeReadJson2(file));
52100
+ if (!entry || entry.instanceId === ours || isPidAlive(entry.pid)) continue;
52101
+ try {
52102
+ fs3.unlinkSync(file);
52103
+ } catch {
52104
+ }
52105
+ }
52106
+ }
52107
+ function registerInstanceAndWarn(entry, warn) {
52108
+ const others = readAllRegistryEntries().filter((e) => e.instanceId !== entry.instanceId && isPidAlive(e.pid));
52109
+ for (const other of others) {
51746
52110
  warn(
51747
52111
  `another bili instance is running (pid ${other.pid}, ${other.origin}) \u2014 both processes will write the same sessions directory; stop one to avoid state pollution (#394)`
51748
52112
  );
51749
52113
  }
51750
- entries.push(entry);
52114
+ reapDeadMarkers(entry.instanceId);
51751
52115
  try {
51752
- fs3.mkdirSync(path7.dirname(file), { recursive: true });
51753
- fs3.writeFileSync(file, JSON.stringify({ instances: entries }) + "\n");
52116
+ atomicWriteJson(entry, registryEntryFile(entry.instanceId));
51754
52117
  } catch {
51755
52118
  }
51756
52119
  }
51757
52120
  function unregisterInstance(instanceId) {
51758
- const file = registryFilePath();
51759
52121
  try {
51760
- const parsed = JSON.parse(fs3.readFileSync(file, "utf8"));
51761
- if (!Array.isArray(parsed.instances)) return;
51762
- const kept = parsed.instances.filter(
51763
- (e) => !(e && typeof e.instanceId === "string" && e.instanceId === instanceId) && isPidAlive(e.pid)
51764
- );
51765
- fs3.writeFileSync(file, JSON.stringify({ instances: kept }) + "\n");
52122
+ fs3.unlinkSync(registryEntryFile(instanceId));
51766
52123
  } catch {
51767
52124
  }
51768
52125
  }
@@ -51879,6 +52236,126 @@ function warnCacheCollapse(session, input, cached) {
51879
52236
  );
51880
52237
  }
51881
52238
 
52239
+ // src/util.ts
52240
+ import { createHash as createHash5 } from "crypto";
52241
+ function hashId2(s3) {
52242
+ return createHash5("sha256").update(s3, "utf8").digest("hex").slice(0, 16);
52243
+ }
52244
+ function isLoopbackAddress(addr) {
52245
+ return !!addr && (addr.startsWith("127.") || addr === "::1" || addr.startsWith("::ffff:127."));
52246
+ }
52247
+ function usageTotals(protocol, usage) {
52248
+ const num3 = (v2) => typeof v2 === "number" && Number.isFinite(v2) ? v2 : void 0;
52249
+ if (protocol === "anthropic") {
52250
+ const fresh = num3(usage["input_tokens"]);
52251
+ const read = num3(usage["cache_read_input_tokens"]);
52252
+ const creation = num3(usage["cache_creation_input_tokens"]);
52253
+ const any = fresh !== void 0 || read !== void 0 || creation !== void 0;
52254
+ return {
52255
+ total: any ? (fresh ?? 0) + (read ?? 0) + (creation ?? 0) : void 0,
52256
+ cached: read
52257
+ };
52258
+ }
52259
+ if (protocol === "openai") {
52260
+ const prompt = num3(usage["prompt_tokens"]);
52261
+ const cached = num3(usage["prompt_tokens_details"]?.["cached_tokens"]);
52262
+ return {
52263
+ total: prompt !== void 0 ? promptInputTotal("openai", prompt, cached) : void 0,
52264
+ cached
52265
+ };
52266
+ }
52267
+ return {
52268
+ total: num3(usage["input_tokens"]),
52269
+ cached: num3(usage["input_tokens_details"]?.["cached_tokens"])
52270
+ };
52271
+ }
52272
+ function promptInputTotal(protocol, input, cached) {
52273
+ if (input === void 0) return 0;
52274
+ const includesCached = protocol === "openai" || protocol === "responses";
52275
+ const splitSemantics = !includesCached || typeof cached === "number" && input < cached;
52276
+ return input + (splitSemantics && typeof cached === "number" ? cached : 0);
52277
+ }
52278
+ function backfillHostUsage(protocol, usage, credit) {
52279
+ if (!Number.isFinite(credit) || credit <= 0) return false;
52280
+ let patched = false;
52281
+ const add = (key) => {
52282
+ if (typeof usage[key] === "number" && Number.isFinite(usage[key])) {
52283
+ usage[key] = usage[key] + credit;
52284
+ patched = true;
52285
+ }
52286
+ };
52287
+ if (protocol === "openai") {
52288
+ add("prompt_tokens");
52289
+ add("total_tokens");
52290
+ } else {
52291
+ add("input_tokens");
52292
+ }
52293
+ return patched;
52294
+ }
52295
+ var CONTEXT_OVERFLOW_PATTERNS = [
52296
+ /context_length_exceeded/i,
52297
+ /context_window_exceeded/i,
52298
+ /context length exceeded/i,
52299
+ /maximum context length/i,
52300
+ /max context length/i,
52301
+ /maximum context size/i,
52302
+ /longer than the model'?s context length/i,
52303
+ /exceeds the context window/i,
52304
+ /out of room in the model/i,
52305
+ /exceeded model token limit/i,
52306
+ /prompt is too long/i,
52307
+ /prompt_too_long/i,
52308
+ /prompt_is_too_long/i,
52309
+ /request_too_large/i,
52310
+ /token limit exceeded/i,
52311
+ // #554: llama.cpp-family "exceed_context_size_error (A / B > W)" — carried by
52312
+ // side requests that bypass preflight; without it the learned channel learns nothing.
52313
+ /exceed[_\s]?context[_\s]?size/i
52314
+ ];
52315
+ function toTokenNumber(s3) {
52316
+ const n = parseInt(s3.replace(/,/g, ""), 10);
52317
+ return Number.isFinite(n) && n >= 1e3 ? n : void 0;
52318
+ }
52319
+ function parseOverflowWindow(text) {
52320
+ let m2 = text.match(/>\s*(\d[\d,]*)\s*maximum/i);
52321
+ if (m2) return toTokenNumber(m2[1]);
52322
+ m2 = text.match(/\(\s*\d[\d,]*\s*\/\s*\d[\d,]*\s*>\s*(\d[\d,]+)\s*\)/);
52323
+ if (m2) return toTokenNumber(m2[1]);
52324
+ m2 = text.match(/maximum context length is (\d[\d,]*)/i) ?? text.match(/maximum context length of (\d[\d,]*)/i) ?? text.match(/maximum context size (?:is|of) (\d[\d,]*)/i) ?? text.match(/(?:maximum|max)\s+(?:context\s+)?length\s+(?:is\s+)?(\d[\d,]*)/i) ?? text.match(/context length\s*\((\d[\d,]*)\s*token/i) ?? text.match(/limit of (\d[\d,]*)\s*token/i) ?? text.match(/(\d[\d,]*)\s*maximum\b/i);
52325
+ if (m2) return toTokenNumber(m2[1]);
52326
+ return void 0;
52327
+ }
52328
+ function inspectContextOverflow(status, bodyText) {
52329
+ const message = (bodyText ?? "").slice(0, 300);
52330
+ if (status !== 400 && status !== 413) return { isOverflow: false, message };
52331
+ if (!bodyText) return { isOverflow: false, message };
52332
+ const isOverflow = CONTEXT_OVERFLOW_PATTERNS.some((p2) => p2.test(bodyText));
52333
+ if (!isOverflow) return { isOverflow: false, message };
52334
+ return { isOverflow: true, window: parseOverflowWindow(bodyText), message };
52335
+ }
52336
+ function reserveOutputHeadroom(window2, maxOutput) {
52337
+ if (Number.isFinite(window2) && window2 > 0 && Number.isFinite(maxOutput) && maxOutput > 0 && maxOutput < window2) {
52338
+ return window2 - maxOutput;
52339
+ }
52340
+ return window2;
52341
+ }
52342
+ function systemToUser(messages) {
52343
+ let hasSys = false;
52344
+ for (const m2 of messages) {
52345
+ if (m2.role === "system" || m2.role === "developer") {
52346
+ hasSys = true;
52347
+ break;
52348
+ }
52349
+ }
52350
+ if (!hasSys) return messages;
52351
+ return messages.map(
52352
+ (m2) => m2.role === "system" || m2.role === "developer" ? { ...m2, role: "user" } : m2
52353
+ );
52354
+ }
52355
+ function shouldReserveOutputHeadroom(protocol) {
52356
+ return protocol !== "anthropic";
52357
+ }
52358
+
51882
52359
  // src/loop/core.ts
51883
52360
  var MAX_LOOP_ROUNDS = 10;
51884
52361
  function isLoopThinking(m2) {
@@ -51932,46 +52409,14 @@ ${lines.join("\n\n")}`;
51932
52409
  }
51933
52410
  return `[Unknown proxy tool: ${toolName}]`;
51934
52411
  }
51935
- function handleAcpStatus(args, ctx) {
51936
- const scope = typeof args.scope === "string" ? args.scope : void 0;
51937
- const view = typeof args.view === "string" ? args.view : void 0;
51938
- const tool = typeof args.tool === "string" ? args.tool : void 0;
51939
- const sort = typeof args.sort === "string" ? args.sort : void 0;
51940
- const limit = typeof args.limit === "number" ? args.limit : void 0;
51941
- const base = buildStatusReport(ctx.session.state, ctx.messages, defaultCountTokens, { scope, view, tool, sort, limit });
51942
- if (scope) return base;
51943
- const nudge = ctx.nudge;
51944
- const ranges = nudge?.compressibleRanges ?? [];
51945
- const protectedRanges = nudge?.protectedRanges ?? [];
51946
- const archive = preCompactionArchiveOf(ctx.session);
51947
- const archivedIds = Object.keys(archive);
51948
- const extra = [];
51949
- if (nudge) {
51950
- extra.push("");
51951
- extra.push(nudge.shouldInject ? `Nudge: ACTIVE \u2014 ${nudge.reason}` : `Nudge: idle \u2014 ${nudge.reason}`);
51952
- }
51953
- if (ranges.length > 0 || protectedRanges.length > 0) {
51954
- extra.push("");
51955
- extra.push(formatRanges(ranges, protectedRanges));
51956
- }
51957
- if (archivedIds.length > 0) {
51958
- extra.push("");
51959
- extra.push(`PRE-COMPACTION ARCHIVE \u2014 ${archivedIds.length} block(s): content was replaced by the client's native compaction summary, so it is no longer in the session history and decompress is unavailable.`);
51960
- for (const id of archivedIds) {
51961
- extra.push(` ${id} \u2014 ${archive[id].reason}`);
51962
- }
51963
- }
51964
- return extra.length > 0 ? `${base}
51965
- ${extra.join("\n")}` : base;
51966
- }
51967
52412
  function recordUsage(ctx, usage, round) {
51968
52413
  const prompt = usage.inputTokens;
51969
52414
  const cached = usage.cachedTokens;
51970
52415
  const out = usage.outputTokens;
51971
- const includesCached = ctx.protocol === "openai" || ctx.protocol === "responses";
51972
- const total = (typeof prompt === "number" ? prompt : 0) + (!includesCached && typeof cached === "number" ? cached : 0);
52416
+ const total = promptInputTotal(ctx.protocol, prompt, cached);
51973
52417
  if (total > 0) ctx.session.stats.inputTokens += total;
51974
52418
  ctx.session.stats.lastInputTokens = Math.max(0, total - (ctx.session.stats.compressCreditTokens ?? 0));
52419
+ ctx.session.hostContextTokens = total + (ctx.session.hostCreditTokens ?? 0);
51975
52420
  if (typeof cached === "number") {
51976
52421
  ctx.session.stats.cachedTokens += cached;
51977
52422
  ctx.session.stats.cacheSamples += 1;
@@ -51995,7 +52440,7 @@ async function* runCompressLoop(upstream, ctx, requestBody, requestOptions, adap
51995
52440
  {
51996
52441
  method: "POST",
51997
52442
  headers: requestOptions.headers,
51998
- body: JSON.stringify(body),
52443
+ body: JSON.stringify(requestOptions.wireTransform ? requestOptions.wireTransform(body) : body),
51999
52444
  ...ctx.proxyUrl ? { dispatcher: proxyDispatcher(ctx.proxyUrl) } : {}
52000
52445
  },
52001
52446
  void 0,
@@ -52110,6 +52555,10 @@ async function* runCompressLoop(upstream, ctx, requestBody, requestOptions, adap
52110
52555
  if (usage.inputTokens !== void 0 || usage.outputTokens !== void 0 || usage.cachedTokens !== void 0) {
52111
52556
  recordUsage(ctx, usage, round);
52112
52557
  }
52558
+ const hostCredit = ctx.session.hostCreditTokens ?? 0;
52559
+ if (hostCredit > 0 && typeof usage.inputTokens === "number") {
52560
+ usage.inputTokens += hostCredit;
52561
+ }
52113
52562
  let resolvedText = assistantText;
52114
52563
  let allCalls = calls;
52115
52564
  if (ctx.textProtocol && assistantText.length > 0 && adapter.extractTextTriggers) {
@@ -52337,96 +52786,6 @@ async function* runCompressLoop(upstream, ctx, requestBody, requestOptions, adap
52337
52786
  }
52338
52787
  }
52339
52788
 
52340
- // src/util.ts
52341
- import { createHash as createHash4 } from "crypto";
52342
- function hashId2(s3) {
52343
- return createHash4("sha256").update(s3, "utf8").digest("hex").slice(0, 16);
52344
- }
52345
- function isLoopbackAddress(addr) {
52346
- return !!addr && (addr.startsWith("127.") || addr === "::1" || addr.startsWith("::ffff:127."));
52347
- }
52348
- function usageTotals(protocol, usage) {
52349
- const num3 = (v2) => typeof v2 === "number" && Number.isFinite(v2) ? v2 : void 0;
52350
- if (protocol === "anthropic") {
52351
- const fresh = num3(usage["input_tokens"]);
52352
- const read = num3(usage["cache_read_input_tokens"]);
52353
- const creation = num3(usage["cache_creation_input_tokens"]);
52354
- const any = fresh !== void 0 || read !== void 0 || creation !== void 0;
52355
- return {
52356
- total: any ? (fresh ?? 0) + (read ?? 0) + (creation ?? 0) : void 0,
52357
- cached: read
52358
- };
52359
- }
52360
- if (protocol === "openai") {
52361
- return {
52362
- total: num3(usage["prompt_tokens"]),
52363
- cached: num3(usage["prompt_tokens_details"]?.["cached_tokens"])
52364
- };
52365
- }
52366
- return {
52367
- total: num3(usage["input_tokens"]),
52368
- cached: num3(usage["input_tokens_details"]?.["cached_tokens"])
52369
- };
52370
- }
52371
- var CONTEXT_OVERFLOW_PATTERNS = [
52372
- /context_length_exceeded/i,
52373
- /context_window_exceeded/i,
52374
- /context length exceeded/i,
52375
- /maximum context length/i,
52376
- /max context length/i,
52377
- /maximum context size/i,
52378
- /longer than the model'?s context length/i,
52379
- /exceeds the context window/i,
52380
- /out of room in the model/i,
52381
- /exceeded model token limit/i,
52382
- /prompt is too long/i,
52383
- /prompt_too_long/i,
52384
- /prompt_is_too_long/i,
52385
- /request_too_large/i,
52386
- /token limit exceeded/i
52387
- ];
52388
- function toTokenNumber(s3) {
52389
- const n = parseInt(s3.replace(/,/g, ""), 10);
52390
- return Number.isFinite(n) && n >= 1e3 ? n : void 0;
52391
- }
52392
- function parseOverflowWindow(text) {
52393
- let m2 = text.match(/>\s*(\d[\d,]*)\s*maximum/i);
52394
- if (m2) return toTokenNumber(m2[1]);
52395
- m2 = text.match(/maximum context length is (\d[\d,]*)/i) ?? text.match(/maximum context length of (\d[\d,]*)/i) ?? text.match(/maximum context size (?:is|of) (\d[\d,]*)/i) ?? text.match(/(?:maximum|max)\s+(?:context\s+)?length\s+(?:is\s+)?(\d[\d,]*)/i) ?? text.match(/context length\s*\((\d[\d,]*)\s*token/i) ?? text.match(/limit of (\d[\d,]*)\s*token/i) ?? text.match(/(\d[\d,]*)\s*maximum\b/i);
52396
- if (m2) return toTokenNumber(m2[1]);
52397
- return void 0;
52398
- }
52399
- function inspectContextOverflow(status, bodyText) {
52400
- const message = (bodyText ?? "").slice(0, 300);
52401
- if (status !== 400 && status !== 413) return { isOverflow: false, message };
52402
- if (!bodyText) return { isOverflow: false, message };
52403
- const isOverflow = CONTEXT_OVERFLOW_PATTERNS.some((p2) => p2.test(bodyText));
52404
- if (!isOverflow) return { isOverflow: false, message };
52405
- return { isOverflow: true, window: parseOverflowWindow(bodyText), message };
52406
- }
52407
- function reserveOutputHeadroom(window2, maxOutput) {
52408
- if (Number.isFinite(window2) && window2 > 0 && Number.isFinite(maxOutput) && maxOutput > 0 && maxOutput < window2) {
52409
- return window2 - maxOutput;
52410
- }
52411
- return window2;
52412
- }
52413
- function systemToUser(messages) {
52414
- let hasSys = false;
52415
- for (const m2 of messages) {
52416
- if (m2.role === "system" || m2.role === "developer") {
52417
- hasSys = true;
52418
- break;
52419
- }
52420
- }
52421
- if (!hasSys) return messages;
52422
- return messages.map(
52423
- (m2) => m2.role === "system" || m2.role === "developer" ? { ...m2, role: "user" } : m2
52424
- );
52425
- }
52426
- function shouldReserveOutputHeadroom(protocol) {
52427
- return protocol !== "anthropic";
52428
- }
52429
-
52430
52789
  // src/loop/adapter-responses.ts
52431
52790
  var RESPONSES_ITEM_ID_MAX = 64;
52432
52791
  function normalizeResponsesMessageItems(input) {
@@ -53058,7 +53417,7 @@ function stripFinishReasonChunk(buf) {
53058
53417
  return buf;
53059
53418
  }
53060
53419
  }
53061
- function createOpenaiAdapter(requestBody, clientSystem) {
53420
+ function createOpenaiAdapter(requestBody, clientSystem, hostCredit = 0) {
53062
53421
  const model = requestBody.model ?? "unknown";
53063
53422
  let responseId = `chatcmpl-proxy-${Date.now()}`;
53064
53423
  let toolIndex = 0;
@@ -53252,7 +53611,24 @@ function createOpenaiAdapter(requestBody, clientSystem) {
53252
53611
  cachedTokens: typeof pd?.cached_tokens === "number" ? pd.cached_tokens : void 0
53253
53612
  };
53254
53613
  if (sawRealToolCall) {
53255
- yield { kind: "meta", chunk: rawBuf };
53614
+ let chunk = rawBuf;
53615
+ if (hostCredit > 0 && u2) {
53616
+ const pu = typeof u2.prompt_tokens === "number" ? u2.prompt_tokens : void 0;
53617
+ const tu = typeof u2.total_tokens === "number" ? u2.total_tokens : void 0;
53618
+ if (pu !== void 0 || tu !== void 0) {
53619
+ const patched = {
53620
+ ...parsed,
53621
+ usage: {
53622
+ ...u2,
53623
+ ...pu !== void 0 ? { prompt_tokens: pu + hostCredit } : {},
53624
+ ...tu !== void 0 ? { total_tokens: tu + hostCredit } : {}
53625
+ }
53626
+ };
53627
+ const out = eventStr.split("\n").map((l) => l.startsWith("data:") ? `data: ${JSON.stringify(patched)}` : l).join("\n");
53628
+ chunk = Buffer.from(out + "\n\n", "utf8");
53629
+ }
53630
+ }
53631
+ yield { kind: "meta", chunk };
53256
53632
  yield { kind: "done", finishReason, suppressCompletion: true };
53257
53633
  continue;
53258
53634
  } else {
@@ -53439,7 +53815,7 @@ data: ${JSON.stringify({ type: "content_block_delta", index, delta: { type: "tex
53439
53815
  "utf8"
53440
53816
  );
53441
53817
  }
53442
- function createAnthropicAdapter(requestBody, originalSystem) {
53818
+ function createAnthropicAdapter(requestBody, originalSystem, hostCredit = 0) {
53443
53819
  const model = requestBody.model ?? void 0;
53444
53820
  let messageId;
53445
53821
  let clientIndex = 0;
@@ -53553,8 +53929,19 @@ ${systemPrompt}` : systemPrompt;
53553
53929
  if (typeof u2.input_tokens === "number") roundInput = u2.input_tokens;
53554
53930
  if (typeof u2.cache_read_input_tokens === "number") roundCached = u2.cache_read_input_tokens;
53555
53931
  if (round === 1) {
53932
+ let chunk = rawBuf;
53933
+ if (hostCredit > 0 && typeof u2.input_tokens === "number") {
53934
+ const patched = structuredClone(data);
53935
+ const pmsg = patched["message"];
53936
+ const pu = pmsg?.["usage"] ?? {};
53937
+ if (typeof pu.input_tokens === "number") {
53938
+ pu.input_tokens += hostCredit;
53939
+ const out = eventStr.split("\n").map((l) => l.startsWith("data:") ? `data: ${JSON.stringify(patched)}` : l).join("\n");
53940
+ chunk = Buffer.from(out + "\n\n", "utf8");
53941
+ }
53942
+ }
53556
53943
  messageStartForwarded = true;
53557
- yield { kind: "meta", chunk: rawBuf, firstRoundOnly: true };
53944
+ yield { kind: "meta", chunk, firstRoundOnly: true };
53558
53945
  }
53559
53946
  } else if (type === "ping") {
53560
53947
  yield { kind: "meta", chunk: rawBuf };
@@ -53737,10 +54124,10 @@ data: ${JSON.stringify({ type: "content_block_stop", index })}
53737
54124
  }
53738
54125
 
53739
54126
  // src/loop/index.ts
53740
- function pickAdapter(protocol, requestBody, textProtocol, responsesProjection, anthropicSystem, openaiSystem) {
54127
+ function pickAdapter(protocol, requestBody, textProtocol, responsesProjection, anthropicSystem, openaiSystem, hostCredit = 0) {
53741
54128
  if (protocol === "responses") return createResponsesAdapter(textProtocol, responsesProjection);
53742
- if (protocol === "openai") return createOpenaiAdapter(requestBody, openaiSystem);
53743
- if (protocol === "anthropic") return createAnthropicAdapter(requestBody, anthropicSystem);
54129
+ if (protocol === "openai") return createOpenaiAdapter(requestBody, openaiSystem, hostCredit);
54130
+ if (protocol === "anthropic") return createAnthropicAdapter(requestBody, anthropicSystem, hostCredit);
53744
54131
  throw new Error(`[acp-loop] unknown protocol: ${protocol}`);
53745
54132
  }
53746
54133
 
@@ -53801,7 +54188,7 @@ function executeProxyTool2(toolName, args, ctx) {
53801
54188
  ${lines.join("\n\n")}`;
53802
54189
  }
53803
54190
  if (toolName === "acp_status") {
53804
- return buildStatusReport(ctx.session.state, ctx.messages, defaultCountTokens);
54191
+ return handleAcpStatus(args, ctx);
53805
54192
  }
53806
54193
  return `[Unknown proxy tool: ${toolName}]`;
53807
54194
  }
@@ -53898,7 +54285,7 @@ async function compressLoopResponsesJson(initialResponse, ctx, requestBody, requ
53898
54285
  const result = await fetchWithRetry(requestOptions.url, {
53899
54286
  method: "POST",
53900
54287
  headers: requestOptions.headers,
53901
- body: JSON.stringify(requestBody),
54288
+ body: JSON.stringify(requestOptions.wireTransform ? requestOptions.wireTransform(requestBody) : requestBody),
53902
54289
  ...ctx.proxyUrl ? { dispatcher: proxyDispatcher(ctx.proxyUrl) } : {}
53903
54290
  }, void 0, void 0, (info) => {
53904
54291
  const lc = lastCompressSuffix(ctx.session.lastCompress);
@@ -54316,6 +54703,35 @@ data: ${JSON.stringify({ type: "message_delta", delta: { stop_reason: "end_turn"
54316
54703
  safeWrite(res, `event: message_stop
54317
54704
  data: ${JSON.stringify({ type: "message_stop" })}
54318
54705
 
54706
+ `);
54707
+ }
54708
+ } catch {
54709
+ } finally {
54710
+ try {
54711
+ res.end();
54712
+ } catch {
54713
+ }
54714
+ }
54715
+ }
54716
+ function emitPreflightError(res, protocol, error, log2) {
54717
+ const err2 = { type: "server_error", code: "preflight_compress_failed", message: error.message, retryable: error.retryable };
54718
+ log2?.(`[acp-proxy: preflight failed after early response commit \u2014 delivering in-band: ${error.message}]`);
54719
+ try {
54720
+ if (protocol === "openai") {
54721
+ safeWrite(res, `data: ${JSON.stringify({ error: err2 })}
54722
+
54723
+ data: [DONE]
54724
+
54725
+ `);
54726
+ } else if (protocol === "responses") {
54727
+ safeWrite(res, `event: error
54728
+ data: ${JSON.stringify({ type: "error", code: err2.code, message: err2.message })}
54729
+
54730
+ `);
54731
+ } else {
54732
+ safeWrite(res, `event: error
54733
+ data: ${JSON.stringify({ type: "error", error: { type: "server_error", code: err2.code, message: err2.message } })}
54734
+
54319
54735
  `);
54320
54736
  }
54321
54737
  } catch {
@@ -54371,7 +54787,7 @@ function codexTurnIdentity(headers) {
54371
54787
  }
54372
54788
 
54373
54789
  // src/prefix-affinity.ts
54374
- import { createHash as createHash5 } from "crypto";
54790
+ import { createHash as createHash6 } from "crypto";
54375
54791
  var MIN_CANONICAL_BYTES = 24;
54376
54792
  var MAX_TRACKED_SESSIONS3 = 256;
54377
54793
  var TTL_MS2 = 7 * 24 * 60 * 60 * 1e3;
@@ -54393,7 +54809,7 @@ function sortKeys2(value) {
54393
54809
  return value;
54394
54810
  }
54395
54811
  function sha256(text) {
54396
- return createHash5("sha256").update(text, "utf8").digest("hex");
54812
+ return createHash6("sha256").update(text, "utf8").digest("hex");
54397
54813
  }
54398
54814
  function hasUserMessage(messages) {
54399
54815
  return messages.some((m2) => !!m2 && typeof m2 === "object" && m2.role === "user");
@@ -55006,7 +55422,7 @@ function handlePluginManifest(res) {
55006
55422
  statusEndpoint: "/__bili/plugin/status"
55007
55423
  }));
55008
55424
  }
55009
- function handlePluginStatus(conversationId2, res, fallbackLatest = false) {
55425
+ function handlePluginStatus(conversationId2, res, deps, fallbackLatest = false) {
55010
55426
  let entry = conversations.get(conversationId2);
55011
55427
  let session = entry ? peekSession(entry.sessionId) : void 0;
55012
55428
  let viaFallback = false;
@@ -55030,16 +55446,33 @@ function handlePluginStatus(conversationId2, res, fallbackLatest = false) {
55030
55446
  const limit = session.metadata.effectiveContextLimit;
55031
55447
  const mem = remembered.get(session.id);
55032
55448
  const modelContextLimit = typeof limit === "number" && limit > 0 ? limit : 0;
55449
+ let nudge;
55450
+ try {
55451
+ const messages = mem ? mem.processed.length > 0 ? mem.processed : mem.original : [];
55452
+ if (messages.length > 0) {
55453
+ nudge = deps.core.processTurn({
55454
+ messages,
55455
+ state: session.state,
55456
+ config: deps.config,
55457
+ tokenCount: session.stats.lastInputTokens,
55458
+ renderTags: "none"
55459
+ }).nudge;
55460
+ }
55461
+ } catch {
55462
+ nudge = void 0;
55463
+ }
55033
55464
  let panel;
55034
55465
  try {
55466
+ const sysTokRaw = session.metadata.systemPromptTokens;
55467
+ const systemPromptTokens = typeof sysTokRaw === "number" && Number.isFinite(sysTokRaw) && sysTokRaw > 0 ? sysTokRaw : 0;
55035
55468
  panel = buildStatusPanel({
55036
55469
  version: `billion-context@${PROXY_VERSION}`,
55037
- tokenCount: session.stats.lastInputTokens,
55038
- systemPromptTokens: 0,
55470
+ tokenCount: session.hostContextTokens ?? session.stats.lastInputTokens,
55471
+ systemPromptTokens,
55039
55472
  state: session.state,
55040
- nudge: mem?.nudge,
55473
+ nudge,
55041
55474
  modelContextLimit,
55042
- unprunedTokens: mem && mem.original.length > 0 ? mem.original.reduce((sum, m2) => sum + defaultCountTokens(m2.text ?? ""), 0) : void 0
55475
+ unprunedTokens: mem && mem.original.length > 0 ? mem.original.reduce((sum, m2) => sum + defaultCountTokens(m2.text ?? ""), 0) + systemPromptTokens : void 0
55043
55476
  });
55044
55477
  } catch {
55045
55478
  panel = void 0;
@@ -55052,7 +55485,8 @@ function handlePluginStatus(conversationId2, res, fallbackLatest = false) {
55052
55485
  label: session.meta.label ?? null,
55053
55486
  pluginAgent: session.metadata.pluginAgent ?? null,
55054
55487
  contextLimit: typeof limit === "number" ? limit : null,
55055
- contextTokens: session.stats.lastInputTokens,
55488
+ contextTokens: session.hostContextTokens ?? session.stats.lastInputTokens,
55489
+ hostCredit: session.hostCreditTokens ?? 0,
55056
55490
  inputTokens: session.stats.inputTokens,
55057
55491
  outputTokens: session.stats.outputTokens,
55058
55492
  cachedTokens: session.stats.cachedTokens,
@@ -55104,8 +55538,7 @@ async function handlePluginTool(payload, res, deps) {
55104
55538
  config: deps.config,
55105
55539
  messages,
55106
55540
  session,
55107
- log: (m2) => deps.log("info", `[${session.id}] [plugin] ${m2}`),
55108
- nudge: mem?.nudge
55541
+ log: (m2) => deps.log("info", `[${session.id}] [plugin] ${m2}`)
55109
55542
  }, callId);
55110
55543
  });
55111
55544
  } catch (err2) {
@@ -55159,16 +55592,16 @@ function usageFromSseEvent(obj) {
55159
55592
  return void 0;
55160
55593
  }
55161
55594
  function applyUsageSample(session, sample, protocol) {
55162
- const includesCached = protocol === "openai" || protocol === "responses";
55163
55595
  if (sample.cachedTokens !== void 0) {
55164
55596
  session.stats.cachedTokens += sample.cachedTokens;
55165
55597
  session.stats.cacheSamples += 1;
55166
55598
  }
55167
55599
  if (sample.inputTokens !== void 0) {
55168
- const total = sample.inputTokens + (!includesCached && sample.cachedTokens !== void 0 ? sample.cachedTokens : 0);
55600
+ const total = promptInputTotal(protocol, sample.inputTokens, sample.cachedTokens);
55169
55601
  session.stats.inputTokens += total;
55170
55602
  session.stats.lastInputTokens = Math.max(0, total - (session.stats.compressCreditTokens ?? 0));
55171
55603
  warnCacheCollapse(session, total, sample.cachedTokens ?? 0);
55604
+ session.hostContextTokens = total + (session.hostCreditTokens ?? 0);
55172
55605
  }
55173
55606
  if (sample.outputTokens !== void 0) session.stats.outputTokens += sample.outputTokens;
55174
55607
  }
@@ -55182,6 +55615,7 @@ async function pipePluginChatWithStrip(stream2, res, protocol, session, log2) {
55182
55615
  const decoder = new TextDecoder("utf-8");
55183
55616
  let buf = "";
55184
55617
  const acc = {};
55618
+ const credit = session?.hostCreditTokens ?? 0;
55185
55619
  const onDrop = (snippet) => {
55186
55620
  log("warn", `[tag-echo] stripped model-emitted render tag (plugin passthrough): ${snippet.slice(0, 80).replace(/\n/g, " ")}`);
55187
55621
  log2?.(`[tag-echo] stripped model-emitted render tag from plugin passthrough text`);
@@ -55339,12 +55773,23 @@ data: ${JSON.stringify({ type: "content_block_delta", index, delta: { type: delt
55339
55773
  if (ev["type"] === "message_stop") sawTerminal = true;
55340
55774
  const sample = usageFromSseEvent(ev);
55341
55775
  if (sample) mergeUsageSample(acc, sample);
55776
+ let backfilled = false;
55777
+ if (credit > 0 && protocol) {
55778
+ const usage = ev["type"] === "message_start" ? ev["message"]?.["usage"] : ev["usage"];
55779
+ const deltaEcho = ev["type"] === "message_delta" && (num2(usage?.["input_tokens"]) ?? 0) <= 0;
55780
+ if (usage && !deltaEcho && backfillHostUsage(protocol, usage, credit)) backfilled = true;
55781
+ }
55342
55782
  const out = protocol === "anthropic" ? processAnthropic(ev, rawEvent) : processOpenai(ev, rawEvent);
55343
- if (out.length > 0) await write(out);
55783
+ if (backfilled && out === rawEvent + "\n\n") {
55784
+ await write(rebuildEvent(rawEvent, ev));
55785
+ } else if (out.length > 0) {
55786
+ await write(out);
55787
+ }
55344
55788
  }
55345
55789
  }
55346
55790
  if (res.destroyed || res.writableEnded) break;
55347
55791
  }
55792
+ if (buf.length > 0 && !res.destroyed && !res.writableEnded) await write(buf);
55348
55793
  const rest = flushTails();
55349
55794
  if (rest.length > 0 && !res.destroyed && !res.writableEnded) await write(rest);
55350
55795
  settleUsage();
@@ -55389,6 +55834,7 @@ async function pipePluginResponsesWithStrip(stream2, res, session, log2) {
55389
55834
  if (!res.write(Buffer.from(s3, "utf8"))) {
55390
55835
  return new Promise((r) => res.once("drain", () => r()));
55391
55836
  }
55837
+ return Promise.resolve();
55392
55838
  };
55393
55839
  const settleUsage = () => {
55394
55840
  if (session && (acc.inputTokens !== void 0 || acc.outputTokens !== void 0 || acc.cachedTokens !== void 0)) {
@@ -55445,7 +55891,17 @@ async function pipePluginResponsesWithStrip(stream2, res, session, log2) {
55445
55891
  const type = ev["type"];
55446
55892
  if (type === "response.output_text.done" || type === "response.content_part.done" || type === "response.output_item.done" || type === "response.completed" || type === "response.failed" || type === "response.incomplete") {
55447
55893
  if (type === "response.completed" || type === "response.failed" || type === "response.incomplete") sawTerminal = true;
55448
- const out = containsRenderTagText(jsonStr) ? rebuildEvent(rawEvent, stripResponsesText(ev)) : rawEvent + "\n\n";
55894
+ let evOut = ev;
55895
+ let rebuild = containsRenderTagText(jsonStr);
55896
+ if (rebuild) evOut = stripResponsesText(ev);
55897
+ if (type === "response.completed") {
55898
+ const credit = session?.hostCreditTokens ?? 0;
55899
+ const usage = evOut["response"]?.["usage"];
55900
+ if (credit > 0 && usage && backfillHostUsage("responses", usage, credit)) {
55901
+ rebuild = true;
55902
+ }
55903
+ }
55904
+ const out = rebuild ? rebuildEvent(rawEvent, evOut) : rawEvent + "\n\n";
55449
55905
  await write(flushTail(out));
55450
55906
  continue;
55451
55907
  }
@@ -55526,8 +55982,10 @@ async function pipePluginJson(stream2, res, session, protocol) {
55526
55982
  }
55527
55983
  reader.releaseLock();
55528
55984
  const text = Buffer.concat(chunks).toString("utf8");
55985
+ let json;
55986
+ let mutated = false;
55529
55987
  try {
55530
- const json = JSON.parse(text);
55988
+ json = JSON.parse(text);
55531
55989
  const usage = json["usage"];
55532
55990
  if (session && usage) {
55533
55991
  const input = num2(usage["prompt_tokens"]) ?? num2(usage["input_tokens"]);
@@ -55538,18 +55996,21 @@ async function pipePluginJson(stream2, res, session, protocol) {
55538
55996
  cachedTokens: num2(usage["prompt_tokens_details"]?.["cached_tokens"]) ?? num2(usage["input_tokens_details"]?.["cached_tokens"]) ?? num2(usage["cache_read_input_tokens"])
55539
55997
  }, protocol);
55540
55998
  markDirty(session);
55999
+ const credit = session.hostCreditTokens ?? 0;
56000
+ if (credit > 0 && protocol && backfillHostUsage(protocol, usage, credit)) {
56001
+ mutated = true;
56002
+ }
55541
56003
  }
55542
56004
  }
55543
56005
  } catch {
55544
56006
  }
55545
- if (containsRenderTagText(text)) {
55546
- try {
55547
- const json = JSON.parse(text);
55548
- const stripped = protocol === "responses" ? stripResponsesText(json) : protocol === "anthropic" ? stripAnthropicText(json) : stripOpenaiChatText(json);
55549
- res.end(Buffer.from(JSON.stringify(stripped), "utf8"));
55550
- return;
55551
- } catch {
55552
- }
56007
+ if (json && containsRenderTagText(text)) {
56008
+ json = protocol === "responses" ? stripResponsesText(json) : protocol === "anthropic" ? stripAnthropicText(json) : stripOpenaiChatText(json);
56009
+ mutated = true;
56010
+ }
56011
+ if (mutated && json) {
56012
+ res.end(Buffer.from(JSON.stringify(json), "utf8"));
56013
+ return;
55553
56014
  }
55554
56015
  res.end(text);
55555
56016
  }
@@ -56223,13 +56684,24 @@ function parseIpLiteral(s3) {
56223
56684
  if (t.includes(":")) return t;
56224
56685
  return null;
56225
56686
  }
56687
+ function normalizeIpLiteral(ip) {
56688
+ const t = ip.trim().toLowerCase();
56689
+ if (!t.startsWith("::ffff:")) return t;
56690
+ const dotted = t.match(/^::ffff:(\d{1,3}(?:\.\d{1,3}){3})$/);
56691
+ if (dotted) return dotted[1];
56692
+ const hex = t.match(/^::ffff:0*([0-9a-f]{1,4}):0*([0-9a-f]{1,4})$/);
56693
+ if (hex) {
56694
+ const hi2 = parseInt(hex[1], 16);
56695
+ const lo2 = parseInt(hex[2], 16);
56696
+ return `${hi2 >> 8 & 255}.${hi2 & 255}.${lo2 >> 8 & 255}.${lo2 & 255}`;
56697
+ }
56698
+ return t;
56699
+ }
56226
56700
  function classifyIp(ip) {
56227
- const lit = parseIpLiteral(ip);
56701
+ const lit = normalizeIpLiteral(parseIpLiteral(ip) ?? "");
56228
56702
  if (!lit) return "public";
56229
56703
  if (lit.includes(":")) {
56230
56704
  if (lit === "::1" || lit === "::") return "loopback";
56231
- const mapped = lit.match(/^::ffff:(\d+\.\d+\.\d+\.\d+)$/);
56232
- if (mapped) return classifyIp(mapped[1]);
56233
56705
  if (lit.startsWith("fe8") || lit.startsWith("fe9") || lit.startsWith("fea") || lit.startsWith("feb")) return "linkLocal";
56234
56706
  if (lit.startsWith("fc") || lit.startsWith("fd")) return "private";
56235
56707
  return "public";
@@ -56280,7 +56752,7 @@ async function checkTunnelDestination(origin, ctx) {
56280
56752
  let ips;
56281
56753
  const literal = parseIpLiteral(host);
56282
56754
  if (literal) {
56283
- ips = [literal];
56755
+ ips = [normalizeIpLiteral(literal)];
56284
56756
  } else {
56285
56757
  try {
56286
56758
  ips = await (ctx.resolveHost ?? dnsResolveHost)(host);
@@ -56291,7 +56763,7 @@ async function checkTunnelDestination(origin, ctx) {
56291
56763
  }
56292
56764
  if (ctx.selfPort !== void 0 && port === ctx.selfPort) {
56293
56765
  const mine = (ctx.localIps ?? localMachineIps)();
56294
- if (ips.some((ip) => mine.has(ip.toLowerCase()) || mine.has(`::ffff:${ip.toLowerCase()}`))) {
56766
+ if (ips.some((ip) => ip === "0.0.0.0" || ip === "::" || mine.has(ip.toLowerCase()) || mine.has(`::ffff:${ip.toLowerCase()}`))) {
56295
56767
  return { ok: false, code: "self", message: "the bili tunnel may not target the proxy itself" };
56296
56768
  }
56297
56769
  }
@@ -57168,10 +57640,7 @@ function resolveUpstream(_opts, reqUrl, req) {
57168
57640
  if (reqUrl.startsWith("http://") || reqUrl.startsWith("https://")) {
57169
57641
  try {
57170
57642
  const u2 = new URL(reqUrl);
57171
- const ownHost = (req?.headers.host ?? "").toLowerCase();
57172
- if (u2.host.toLowerCase() !== ownHost) {
57173
- return { upstream: `${u2.protocol}//${u2.host}`, rewrittenUrl: reqUrl, tunnel: true };
57174
- }
57643
+ return { upstream: `${u2.protocol}//${u2.host}`, rewrittenUrl: reqUrl, tunnel: true };
57175
57644
  } catch {
57176
57645
  }
57177
57646
  }
@@ -57387,6 +57856,16 @@ function restoreOutputBudget(parsed, session, log2) {
57387
57856
  log2("info", `[${session.id}] output budget restored ${value} -> ${highWater} (#546: client shrank it from its raw-history estimate)`);
57388
57857
  }
57389
57858
  }
57859
+ var SIDE_REQUEST_GUARD_TOLERANCE = 1.15;
57860
+ function sideRequestGuard(parsed, protocol, modelContextLimit, learnedLimit) {
57861
+ let limit = modelContextLimit;
57862
+ if (typeof learnedLimit === "number" && learnedLimit > 0 && learnedLimit < limit) limit = learnedLimit;
57863
+ const field = outputBudgetField(parsed);
57864
+ const maxOut = field ? parsed[field] : 0;
57865
+ if (limit > 0 && shouldReserveOutputHeadroom(protocol)) limit = reserveOutputHeadroom(limit, maxOut);
57866
+ const estimate = estimateRawBodyTokens(parsed) + imageTokensInParsedBody(protocol, parsed);
57867
+ return { blocked: limit > 0 && estimate >= limit * SIDE_REQUEST_GUARD_TOLERANCE, estimate, limit };
57868
+ }
57390
57869
  function isTrustedAdminOrigin(origin, host, trustedHosts) {
57391
57870
  if (!host || !trustedHosts.has(host.toLowerCase())) return false;
57392
57871
  if (!origin) return true;
@@ -57468,6 +57947,7 @@ async function handle(req, res, opts, core, config, log2, instanceId, instanceSt
57468
57947
  opts.proxySource = fresh.proxySource;
57469
57948
  opts.proxyFallback = fresh.proxyFallback;
57470
57949
  opts.compress = fresh.compress;
57950
+ opts.compat = fresh.compat;
57471
57951
  resetProxyCache();
57472
57952
  for (const k2 of Object.keys(opts.routes)) delete opts.routes[k2];
57473
57953
  Object.assign(opts.routes, loadRoutes());
@@ -57508,7 +57988,7 @@ async function handle(req, res, opts, core, config, log2, instanceId, instanceSt
57508
57988
  try {
57509
57989
  const result = await fetchWithTimeout(targetUrl2, {
57510
57990
  method: "HEAD",
57511
- ...proxyUrl ? { dispatcher: proxyDispatcher(proxyUrl) } : {}
57991
+ ...proxyUrl ? { dispatcher: proxyDispatcher(proxyUrl, 15e3) } : {}
57512
57992
  }, 15e3);
57513
57993
  result.clearTimer();
57514
57994
  recordUpstreamConnection(targetUrl2, proxyUrl);
@@ -57531,7 +58011,7 @@ async function handle(req, res, opts, core, config, log2, instanceId, instanceSt
57531
58011
  res.end(JSON.stringify({ ok: false, error: "conversationId query parameter is required" }));
57532
58012
  return;
57533
58013
  }
57534
- return handlePluginStatus(conversationId2, res, params.get("fallback") === "latest");
58014
+ return handlePluginStatus(conversationId2, res, { core, config, log: log2 }, params.get("fallback") === "latest");
57535
58015
  }
57536
58016
  if (req.method === "POST" && req.url === "/__bili/plugin/tool") {
57537
58017
  try {
@@ -57787,6 +58267,25 @@ ${bodyBuffer.toString("utf8")}`);
57787
58267
  const pluginMode = pluginAgent !== void 0;
57788
58268
  restoreOutputBudget(parsed, session, log2);
57789
58269
  if (!countTokens && !responsesCompact && protocol !== null && isSideRequest(parsed)) {
58270
+ const reqModel2 = parsed.model;
58271
+ const learnedMap2 = session.metadata.learnedContextLimits;
58272
+ const learnedLimit2 = (reqModel2 && learnedMap2 ? learnedMap2[reqModel2] : void 0) ?? session.metadata.learnedContextLimit;
58273
+ const guard = sideRequestGuard(parsed, protocol, reqConfig.modelContextLimit, learnedLimit2);
58274
+ if (guard.blocked) {
58275
+ log2("warn", `[${session.id}] side request (~${guard.estimate} tokens) \u2265 effective window ${guard.limit} (model=${reqModel2 ?? "?"}) \u2014 NOT forwarded: guaranteed upstream 400 (side requests bypass preflight by design, #388)`);
58276
+ if (!res.headersSent && !res.writableEnded && !res.destroyed) {
58277
+ res.writeHead(413, { "content-type": "application/json" });
58278
+ res.end(JSON.stringify({
58279
+ error: {
58280
+ type: "server_error",
58281
+ code: "side_request_payload_too_large",
58282
+ message: `side request payload ~${guard.estimate} tokens reaches the effective context window ${guard.limit} (model=${reqModel2 ?? "unknown"}); NOT forwarded \u2014 side requests (max_tokens<=${SIDE_REQUEST_MAX_TOKENS}) bypass compression by design (#388). Shrink the conversation or raise the model's context window.`,
58283
+ retryable: false
58284
+ }
58285
+ }));
58286
+ }
58287
+ return;
58288
+ }
57790
58289
  log2("info", `[${session.id}] side request (max_tokens<=${SIDE_REQUEST_MAX_TOKENS}) \u2192 passthrough + tag strip only, kernel state untouched`);
57791
58290
  const sidePrepared = {
57792
58291
  body: bodyBuffer,
@@ -57876,19 +58375,37 @@ ${bodyBuffer.toString("utf8")}`);
57876
58375
  instanceId
57877
58376
  );
57878
58377
  if (isPreflightFailFast(outcome)) {
57879
- if (outcome.respond && !res.headersSent && !res.destroyed) {
57880
- res.writeHead(outcome.status, {
57881
- "content-type": "application/json",
57882
- ...outcome.status === 503 ? { "retry-after": "30" } : {}
57883
- });
57884
- res.end(JSON.stringify({
57885
- error: {
57886
- type: "server_error",
57887
- code: "preflight_compress_failed",
57888
- message: outcome.message,
57889
- retryable: outcome.retryable
58378
+ if (outcome.respond && !res.destroyed) {
58379
+ if (res.headersSent) {
58380
+ if (prepared.stream) {
58381
+ emitPreflightError(res, prepared.protocol, { message: outcome.message, retryable: outcome.retryable }, (m2) => log2("warn", m2));
58382
+ } else {
58383
+ try {
58384
+ res.end(JSON.stringify({
58385
+ error: {
58386
+ type: "server_error",
58387
+ code: "preflight_compress_failed",
58388
+ message: outcome.message,
58389
+ retryable: outcome.retryable
58390
+ }
58391
+ }));
58392
+ } catch {
58393
+ }
57890
58394
  }
57891
- }));
58395
+ } else {
58396
+ res.writeHead(outcome.status, {
58397
+ "content-type": "application/json",
58398
+ ...outcome.status === 503 ? { "retry-after": "30" } : {}
58399
+ });
58400
+ res.end(JSON.stringify({
58401
+ error: {
58402
+ type: "server_error",
58403
+ code: "preflight_compress_failed",
58404
+ message: outcome.message,
58405
+ retryable: outcome.retryable
58406
+ }
58407
+ }));
58408
+ }
57892
58409
  }
57893
58410
  return;
57894
58411
  }
@@ -57921,6 +58438,59 @@ function stripKernelSummaries(messages, state) {
57921
58438
  }
57922
58439
  return messages.filter((m2) => !(m2.id ?? "").startsWith("acp_summary_") || !carried.has(m2.id));
57923
58440
  }
58441
+ var RESPONSES_TURN_SEPARATOR = "[The exchange between these two assistant turns was compressed.]";
58442
+ function repairResponsesAssistantOrdering(folded, original) {
58443
+ const runOf = /* @__PURE__ */ new Map();
58444
+ const runHasBody = /* @__PURE__ */ new Map();
58445
+ let run = 0;
58446
+ let inRun = false;
58447
+ for (const m2 of original) {
58448
+ if (m2.role === "assistant") {
58449
+ if (!inRun) {
58450
+ run++;
58451
+ inRun = true;
58452
+ }
58453
+ runOf.set(m2.id, run);
58454
+ if (m2.contentType !== "reasoning") runHasBody.set(run, true);
58455
+ } else {
58456
+ inRun = false;
58457
+ }
58458
+ }
58459
+ const survivorCount = /* @__PURE__ */ new Map();
58460
+ for (const m2 of folded) {
58461
+ const r = m2.role === "assistant" ? runOf.get(m2.id) : void 0;
58462
+ if (r !== void 0) survivorCount.set(r, (survivorCount.get(r) ?? 0) + 1);
58463
+ }
58464
+ const out = [];
58465
+ let phase = -1;
58466
+ let seenReasoning = false;
58467
+ let sepSeq = 0;
58468
+ const pushSeparator = () => {
58469
+ sepSeq++;
58470
+ out.push({ id: `acp_turn_sep_${sepSeq}`, role: "user", contentType: "text", text: RESPONSES_TURN_SEPARATOR });
58471
+ phase = -1;
58472
+ seenReasoning = false;
58473
+ };
58474
+ for (const m2 of folded) {
58475
+ if (m2.role !== "assistant") {
58476
+ out.push(m2);
58477
+ phase = -1;
58478
+ seenReasoning = false;
58479
+ continue;
58480
+ }
58481
+ const kind = m2.contentType === "reasoning" ? "reasoning" : m2.contentType === "tool-call" ? "tool-call" : "message";
58482
+ const r = runOf.get(m2.id);
58483
+ if (kind === "reasoning" && r !== void 0 && runHasBody.get(r) && survivorCount.get(r) === 1) continue;
58484
+ if (kind === "reasoning" && (phase > 0 || seenReasoning) || kind === "message" && phase === 2) pushSeparator();
58485
+ out.push(m2);
58486
+ if (kind === "reasoning") {
58487
+ phase = Math.max(phase, 0);
58488
+ seenReasoning = true;
58489
+ } else if (kind === "message") phase = Math.max(phase, 1);
58490
+ else phase = Math.max(phase, 2);
58491
+ }
58492
+ return out;
58493
+ }
57924
58494
  function diagTagSummary(messages, sessionId, strategy) {
57925
58495
  let textTagged = 0;
57926
58496
  let toolTagged = 0;
@@ -57946,10 +58516,19 @@ function diagNudge(turn, sessionId, tokenCount, limit, model, willInject) {
57946
58516
  const modelTag = model ? ` model=${model}` : "";
57947
58517
  return `[${sessionId}] nudge ${inject}: usage=${pct2} (${tokenCount}/${limit}), growth=${growth}/${floor} (ref=${ref}, interval=${interval}), pendingT1=${pendingT1}/${interval}${modelTag}, reason="${n.reason.slice(0, 120)}"`;
57948
58518
  }
58519
+ function armHostUsageCredit(session, originalMessages, processedMessages, log2) {
58520
+ session.hostCreditTokens = 0;
58521
+ if (session.metadata.pluginAgent === "pi") return;
58522
+ session.hostCreditTokens = processedMessages.length > 0 ? Math.max(0, estimateCoreMessages(originalMessages) - estimateCoreMessages(processedMessages)) : 0;
58523
+ if (session.hostCreditTokens > 0) {
58524
+ log2("info", `[${session.id}] host usage backfill armed: +${session.hostCreditTokens} tok (forwarded view is folded); host usage will report the uncompressed baseline`);
58525
+ }
58526
+ }
57949
58527
  function prepareAnthropic(parsed, req, opts, core, config, prompts, log2, session, pluginMode) {
57950
58528
  const sessionId = session.id;
57951
58529
  const stream2 = parsed.stream === true;
57952
58530
  ++session.stats.requests;
58531
+ session.hostCreditTokens = 0;
57953
58532
  const injectTools = opts.compress.injectTool && !pluginMode;
57954
58533
  if (isAutoModeClassifier(parsed)) {
57955
58534
  log2("info", `[${sessionId}] auto-mode classifier passthrough (skipping compress injection)`);
@@ -58005,20 +58584,46 @@ function prepareAnthropic(parsed, req, opts, core, config, prompts, log2, sessio
58005
58584
  log2("warn", `[${sessionId}] kernel transform failed, forwarding unchanged: ${String(err2)}`);
58006
58585
  processedMessages = [];
58007
58586
  }
58587
+ session.metadata.systemPromptTokens = countSystemAndToolsTokens(extractSystem(systemOut), toolsOut);
58008
58588
  snapshotMessages(session, originalMessages);
58009
58589
  markDirty(session);
58010
58590
  const rebuilt = { ...parsed, messages: rebuiltMessages, system: systemOut, tools: toolsOut };
58011
58591
  delete rebuilt.prompt_cache_key;
58592
+ armHostUsageCredit(session, originalMessages, processedMessages, log2);
58012
58593
  return { body: JSON.stringify(rebuilt), session, processedMessages, originalMessages, anthropicSystem: parsed.system, protocol: "anthropic", stream: stream2, compressInjected: injectTools, pluginMode, nudge, prompts, renderTags: "text-only" };
58013
58594
  }
58014
58595
  var OUTPUT_CLAMP_MARGIN_PCT = 0.05;
58015
58596
  var OUTPUT_CLAMP_MIN_MARGIN = 2048;
58016
58597
  var OUTPUT_CLAMP_FLOOR = 1024;
58017
58598
  var EMERGENCY_NUDGE_ESCALATION_PCT = 0.7;
58599
+ function countSystemAndToolsTokens(systemText, tools) {
58600
+ return defaultCountTokens(systemText ?? "") + defaultCountTokens(JSON.stringify(tools ?? []));
58601
+ }
58018
58602
  function estimateInputTokens(processedMessages, systemText, tools, lastInputTokens) {
58019
- const est = estimateCoreMessages(processedMessages) + defaultCountTokens(systemText ?? "") + defaultCountTokens(JSON.stringify(tools ?? []));
58603
+ const est = estimateCoreMessages(processedMessages) + countSystemAndToolsTokens(systemText, tools);
58020
58604
  return Math.max(lastInputTokens > 0 ? lastInputTokens : 0, est);
58021
58605
  }
58606
+ function estimateWireOverhead(protocol, body) {
58607
+ let parsed;
58608
+ try {
58609
+ parsed = JSON.parse(typeof body === "string" ? body : body.toString("utf8"));
58610
+ } catch {
58611
+ return 0;
58612
+ }
58613
+ const sysRaw = protocol === "responses" ? parsed.instructions : parsed.system;
58614
+ let sysText = "";
58615
+ if (typeof sysRaw === "string") {
58616
+ sysText = sysRaw;
58617
+ } else if (Array.isArray(sysRaw)) {
58618
+ sysText = sysRaw.map((part) => typeof part?.text === "string" ? part.text : "").join("\n");
58619
+ }
58620
+ if (protocol === "openai" && Array.isArray(parsed.messages)) {
58621
+ const hoisted = parsed.messages.filter((m2) => m2.role === "system" || m2.role === "developer").map((m2) => typeof m2.content === "string" ? m2.content : "").join("\n");
58622
+ sysText = sysText ? `${sysText}
58623
+ ${hoisted}` : hoisted;
58624
+ }
58625
+ return defaultCountTokens(sysText) + defaultCountTokens(JSON.stringify(parsed.tools ?? []));
58626
+ }
58022
58627
  function clampOutputBudget(requested, inputEstimate, nativeWindow) {
58023
58628
  const margin = Math.max(OUTPUT_CLAMP_MIN_MARGIN, Math.ceil(inputEstimate * OUTPUT_CLAMP_MARGIN_PCT));
58024
58629
  const cap = nativeWindow - inputEstimate - margin;
@@ -58044,7 +58649,9 @@ function prepareOpenai(parsed, req, opts, core, config, prompts, log2, session,
58044
58649
  const sessionId = session.id;
58045
58650
  const stream2 = parsed.stream === true;
58046
58651
  ++session.stats.requests;
58652
+ session.hostCreditTokens = 0;
58047
58653
  let openaiSystemText = "";
58654
+ let openaiOutboundSystem;
58048
58655
  let processedMessages = [];
58049
58656
  let originalMessages = [];
58050
58657
  let nudge;
@@ -58085,6 +58692,7 @@ function prepareOpenai(parsed, req, opts, core, config, prompts, log2, session,
58085
58692
  if (systemText) sysParts.push(systemText);
58086
58693
  if (shouldInject) sysParts.push(buildCompressSystemPrompt(prompts));
58087
58694
  rebuiltMessages = injectOpenaiSystem(rebuiltMessages, sysParts);
58695
+ openaiOutboundSystem = sysParts.join("\n\n");
58088
58696
  if (injectTools) {
58089
58697
  toolsOut = injectOpenaiTool(parsed.tools);
58090
58698
  }
@@ -58107,6 +58715,10 @@ function prepareOpenai(parsed, req, opts, core, config, prompts, log2, session,
58107
58715
  if (stream2 && rebuilt.stream_options === void 0) {
58108
58716
  rebuilt.stream_options = { include_usage: true };
58109
58717
  }
58718
+ armHostUsageCredit(session, originalMessages, processedMessages, log2);
58719
+ if (!isTitleGen && openaiOutboundSystem !== void 0) {
58720
+ session.metadata.systemPromptTokens = countSystemAndToolsTokens(openaiOutboundSystem, toolsOut);
58721
+ }
58110
58722
  snapshotMessages(session, originalMessages);
58111
58723
  markDirty(session);
58112
58724
  return { body: JSON.stringify(rebuilt), session, processedMessages, originalMessages, protocol: "openai", stream: stream2, compressInjected: injectTools, pluginMode, nudge, prompts, openaiSystemText, renderTags: "text-only" };
@@ -58115,6 +58727,7 @@ function prepareResponses(parsed, req, opts, core, config, prompts, log2, sessio
58115
58727
  const sessionId = session.id;
58116
58728
  const stream2 = parsed.stream === true;
58117
58729
  ++session.stats.requests;
58730
+ session.hostCreditTokens = 0;
58118
58731
  if (reconcileNativeCompactionBoundary(session)) {
58119
58732
  log2("info", `[${sessionId}] reconciled ACP state after native Responses compact boundary`);
58120
58733
  }
@@ -58134,6 +58747,7 @@ function prepareResponses(parsed, req, opts, core, config, prompts, log2, sessio
58134
58747
  let rebuiltInput = parsed.input;
58135
58748
  let toolsOut = parsed.tools;
58136
58749
  let transformOk = false;
58750
+ let responsesDevContent;
58137
58751
  const typedItems = normalizeResponsesMessageItems(parsed.input);
58138
58752
  if (typedItems > 0) {
58139
58753
  log2("info", `[${sessionId}] stamped type:"message" on ${typedItems} type-less input item(s) before projection (omp wire form)`);
@@ -58174,19 +58788,21 @@ function prepareResponses(parsed, req, opts, core, config, prompts, log2, sessio
58174
58788
  log2("info", diagTagSummary(turn.messages, sessionId, "text-only"));
58175
58789
  const willInjectNudge = opts.compress.injectNudge && !!turn.nudge && shouldInject && !isCompactionTrigger && (turn.nudge.shouldInject || emergencyNudge(turn.nudge));
58176
58790
  log2("info", diagNudge(turn, sessionId, tokenCount, config.modelContextLimit, parsed.model, willInjectNudge));
58177
- processedMessages = stripKernelSummaries(turn.messages, turn.state);
58791
+ processedMessages = repairResponsesAssistantOrdering(stripKernelSummaries(turn.messages, turn.state), originalMessages);
58178
58792
  reapOrphanBlocks(session, msgs, deactivateBlock);
58179
58793
  rebuiltInput = patchResponsesInput(projection, processedMessages);
58180
58794
  const forgedSummaries = echoReplaced ? [] : session.metadata.codexForgedSummaries ?? [];
58181
58795
  if (shouldInject && !isCompactionTrigger && !process.env.ACP_NO_COMPRESS_PROMPT) {
58182
58796
  const prompt = responsesTextProtocol ? buildCompressHybridSystemPrompt(prompts) : buildCompressSystemPrompt(prompts);
58183
58797
  const devContent = [...projection.systemParts, ...forgedSummaries, prompt].join("\n\n---\n\n");
58798
+ responsesDevContent = devContent;
58184
58799
  rebuiltInput = injectResponsesDeveloperMessage(rebuiltInput, devContent);
58185
58800
  if (!process.env.ACP_NO_INJECT_TOOL && injectTools) {
58186
58801
  toolsOut = responsesTextProtocol ? injectResponsesTool(parsed.tools, ACP_READONLY_TOOLS_RESPONSES) : injectResponsesTool(parsed.tools);
58187
58802
  }
58188
58803
  } else if (projection.systemParts.length > 0 || forgedSummaries.length > 0) {
58189
58804
  const devContent = [...projection.systemParts, ...forgedSummaries].join("\n\n---\n\n");
58805
+ responsesDevContent = devContent;
58190
58806
  rebuiltInput = injectResponsesDeveloperMessage(rebuiltInput, devContent);
58191
58807
  }
58192
58808
  if (willInjectNudge && turn.nudge) {
@@ -58244,6 +58860,10 @@ function prepareResponses(parsed, req, opts, core, config, prompts, log2, sessio
58244
58860
  });
58245
58861
  log2("info", `[${sessionId}] responses forward tools=[${fwdTools.join(",")}] injectTool=${injectTools}${pluginMode ? " (plugin mode: wire injection suppressed)" : ""} NO_INJECT_TOOL=${!!process.env.ACP_NO_INJECT_TOOL} NO_COMPRESS_PROMPT=${!!process.env.ACP_NO_COMPRESS_PROMPT}`);
58246
58862
  }
58863
+ armHostUsageCredit(session, originalMessages, processedMessages, log2);
58864
+ if (transformOk) {
58865
+ session.metadata.systemPromptTokens = countSystemAndToolsTokens(responsesDevContent ?? "", toolsOut);
58866
+ }
58247
58867
  snapshotMessages(session, originalMessages);
58248
58868
  markDirty(session);
58249
58869
  return {
@@ -58330,7 +58950,7 @@ function prepareResponsesCompact(body, parsed, session, req, core, config, log2)
58330
58950
  session.state = prevState;
58331
58951
  return base;
58332
58952
  }
58333
- const processed = stripKernelSummaries(turn.messages, turn.state);
58953
+ const processed = repairResponsesAssistantOrdering(stripKernelSummaries(turn.messages, turn.state), projection.msgs);
58334
58954
  const output = patchResponsesInput(projection, processed);
58335
58955
  if (typeof output === "string") {
58336
58956
  session.state = prevState;
@@ -58415,6 +59035,12 @@ function logUpstreamProxyDecision(opts, upstreamUrl, decision) {
58415
59035
  const via = decision.proxy ? `via ${maskUrlForLog(decision.proxy)}` : "direct";
58416
59036
  logMsg(opts, "info", `[upstream-proxy] ${maskHostPortForLog(host)} ${via} (source=${decision.source})`);
58417
59037
  }
59038
+ function inferWireProtocol(path18) {
59039
+ const p2 = path18.split("?", 2)[0];
59040
+ if (p2.endsWith("/chat/completions")) return "openai";
59041
+ if (p2.endsWith("/responses") || p2.endsWith("/responses/compact")) return "responses";
59042
+ return null;
59043
+ }
58418
59044
  function buildForwardTarget(req, opts, route, affinity, hopMarker) {
58419
59045
  const reqUrl = req.url ?? "";
58420
59046
  const isAbsoluteUrl = /^https?:\/\//i.test(reqUrl);
@@ -58446,12 +59072,53 @@ function buildForwardTarget(req, opts, route, affinity, hopMarker) {
58446
59072
  function isPreflightFailFast(outcome) {
58447
59073
  return "failFast" in outcome;
58448
59074
  }
59075
+ var PREFLIGHT_HOLD_GRACE_DEFAULT_MS = 3e4;
59076
+ var PREFLIGHT_KEEPALIVE_MS = 15e3;
59077
+ function preflightHoldGraceMs() {
59078
+ const raw = process.env.BILI_PREFLIGHT_HOLD_MS;
59079
+ if (!raw) return PREFLIGHT_HOLD_GRACE_DEFAULT_MS;
59080
+ const v2 = Number(raw);
59081
+ return Number.isFinite(v2) && v2 >= 0 ? Math.floor(v2) : PREFLIGHT_HOLD_GRACE_DEFAULT_MS;
59082
+ }
59083
+ function beginPreflightHold(res, prepared, log2) {
59084
+ if (res.headersSent || res.destroyed || res.writableEnded) return void 0;
59085
+ const sid = prepared.session.id;
59086
+ const keepAlive = prepared.stream ? ": bili-preflight\n\n" : " ";
59087
+ try {
59088
+ if (prepared.stream) {
59089
+ res.writeHead(200, {
59090
+ "content-type": "text/event-stream",
59091
+ "cache-control": "no-cache",
59092
+ "x-accel-buffering": "no",
59093
+ "x-bili-preflight": "compressing"
59094
+ });
59095
+ } else {
59096
+ res.writeHead(200, { "content-type": "application/json", "x-bili-preflight": "compressing" });
59097
+ }
59098
+ } catch {
59099
+ return void 0;
59100
+ }
59101
+ log2("info", `[${sid}] preflight still running after ${preflightHoldGraceMs()}ms grace \u2014 committed early ${prepared.stream ? "SSE" : "JSON"} headers + keep-alive to hold the client (#568)`);
59102
+ try {
59103
+ res.write(keepAlive);
59104
+ } catch {
59105
+ }
59106
+ const iv = setInterval(() => {
59107
+ try {
59108
+ res.write(keepAlive);
59109
+ } catch {
59110
+ clearInterval(iv);
59111
+ }
59112
+ }, PREFLIGHT_KEEPALIVE_MS);
59113
+ return () => clearInterval(iv);
59114
+ }
58449
59115
  async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, core, config, model, route, affinity, log2, instanceId) {
58450
59116
  const session = prepared.session;
58451
59117
  const limit = config.modelContextLimit;
58452
59118
  const imageTokens = imageTokensInRawBody(prepared.protocol, prepared.body);
58453
59119
  const textEstimate = estimateCoreMessages(prepared.processedMessages);
58454
- const payloadEstimate = textEstimate + imageTokens;
59120
+ const overheadEstimate = estimateWireOverhead(prepared.protocol, prepared.body);
59121
+ const payloadEstimate = textEstimate + overheadEstimate + imageTokens;
58455
59122
  const tokenCount = Math.max(session.stats.lastInputTokens, payloadEstimate);
58456
59123
  if (limit <= 0 || !model || tokenCount < limit) return prepared;
58457
59124
  const learnedMap = session.metadata.learnedContextLimits;
@@ -58467,12 +59134,9 @@ async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, c
58467
59134
  log2("error", `[${session.id}] preflight fail-fast ${status2} (retryable=${retryable}): ${message}`);
58468
59135
  return { failFast: true, status: status2, message, retryable, respond: !res.writableEnded };
58469
59136
  };
58470
- if ((prepared.nudge?.compressibleRanges ?? []).length === 0) {
58471
- if (payloadEstimate < limit) {
58472
- log2("warn", `[${session.id}] preflight trigger fired on a stale baseline (~${tokenCount}) but the payload fits (~${payloadEstimate}/${limit}); forwarding as-is`);
58473
- return prepared;
58474
- }
58475
- return failFast(502, "no part of the conversation is compressible (nothing left to fold)", false);
59137
+ if ((prepared.nudge?.compressibleRanges ?? []).length === 0 && payloadEstimate < limit) {
59138
+ log2("warn", `[${session.id}] preflight trigger fired on a stale baseline (~${tokenCount}) but the payload fits (~${payloadEstimate}/${limit}); forwarding as-is`);
59139
+ return prepared;
58476
59140
  }
58477
59141
  log2("warn", `[${session.id}] context ${tokenCount} tokens exceeds model window ${limit} (model=${model}); preflight compressing before forward`);
58478
59142
  const { upstreamUrl, headers, proxyUrl } = buildForwardTarget(req, opts, route, affinity, instanceId);
@@ -58481,31 +59145,42 @@ async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, c
58481
59145
  if (!res.writableEnded) clientAbort.abort();
58482
59146
  });
58483
59147
  const started = Date.now();
58484
- const result = await preflightCompress(
58485
- {
58486
- core,
58487
- session,
58488
- config,
58489
- prompts: prepared.prompts ?? defaultPrompts,
58490
- protocol: prepared.protocol,
58491
- url: upstreamUrl,
58492
- headers,
58493
- model,
58494
- proxyUrl,
58495
- signal: clientAbort.signal,
58496
- log: log2,
58497
- imageFloor: imageTokens
58498
- },
58499
- prepared.originalMessages
58500
- );
58501
- if (result.payloadEstimate < limit) {
58502
- if (result.compressedRanges > 0) {
58503
- log2("info", `[${session.id}] preflight compressed ${result.compressedRanges} range(s), ~${result.savedTokens} tokens saved (${tokenCount} \u2192 ${session.stats.lastInputTokens}) in ${Date.now() - started}ms; rebuilding payload`);
58504
- const rebuilt = runPrepare();
58505
- session.stats.requests -= 1;
58506
- return rebuilt;
58507
- }
58508
- log2("warn", `[${session.id}] preflight made no progress but the payload fits (~${result.payloadEstimate}/${limit}); forwarding as-is`);
59148
+ let stopHold;
59149
+ const holdTimer = setTimeout(() => {
59150
+ stopHold = beginPreflightHold(res, prepared, log2);
59151
+ }, preflightHoldGraceMs());
59152
+ holdTimer.unref();
59153
+ let result;
59154
+ try {
59155
+ result = await preflightCompress(
59156
+ {
59157
+ core,
59158
+ session,
59159
+ config,
59160
+ prompts: prepared.prompts ?? defaultPrompts,
59161
+ protocol: prepared.protocol,
59162
+ url: upstreamUrl,
59163
+ headers,
59164
+ model,
59165
+ proxyUrl,
59166
+ signal: clientAbort.signal,
59167
+ log: log2,
59168
+ imageFloor: imageTokens,
59169
+ wireOverhead: overheadEstimate
59170
+ },
59171
+ prepared.originalMessages
59172
+ );
59173
+ } finally {
59174
+ clearTimeout(holdTimer);
59175
+ stopHold?.();
59176
+ }
59177
+ if (result.compressedRanges > 0) {
59178
+ log2("info", `[${session.id}] preflight compressed ${result.compressedRanges} range(s), ~${result.savedTokens} tokens saved (${tokenCount} \u2192 ${session.stats.lastInputTokens}) in ${Date.now() - started}ms; rebuilding payload`);
59179
+ const rebuilt = runPrepare();
59180
+ session.stats.requests -= 1;
59181
+ if (estimateCoreMessages(rebuilt.processedMessages) + overheadEstimate + imageTokens < limit) return rebuilt;
59182
+ } else if (estimateCoreMessages(prepared.processedMessages) + overheadEstimate + imageTokens < limit) {
59183
+ log2("warn", `[${session.id}] preflight made no progress but the payload fits; forwarding as-is`);
58509
59184
  return prepared;
58510
59185
  }
58511
59186
  const f2 = result.failure;
@@ -58514,16 +59189,40 @@ async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, c
58514
59189
  return { failFast: true, status: 0, message: f2.detail, retryable: false, respond: false };
58515
59190
  }
58516
59191
  const status = f2?.kind === "upstream" && f2.status === 429 ? 503 : 502;
58517
- return failFast(status, f2?.detail ?? "unknown reason", status === 503);
59192
+ return failFast(status, f2?.detail ?? "the payload still exceeds the window after preflight compression", status === 503);
58518
59193
  }
58519
59194
  async function forward(req, res, opts, body, prepared, core, config, log2, route, instanceId, affinity) {
58520
59195
  if (prepared?.codexForge) {
58521
59196
  log2("info", `[${prepared.session.id}] codex compact served locally (${prepared.codexForge.kind}); upstream not contacted`);
58522
- res.writeHead(200, { "content-type": prepared.codexForge.contentType });
59197
+ if (!res.headersSent) res.writeHead(200, { "content-type": prepared.codexForge.contentType });
58523
59198
  res.end(prepared.codexForge.body);
58524
59199
  return;
58525
59200
  }
59201
+ let wireBody = body;
59202
+ let compatRoles = null;
59203
+ let compatProtocol = null;
58526
59204
  const { upstreamUrl, headers, proxyUrl } = buildForwardTarget(req, opts, route, affinity, prepared !== null ? instanceId : void 0);
59205
+ if (typeof body === "string") {
59206
+ const configured = resolveCompatRoles(opts.routes, upstreamUrl, opts.compat?.roles);
59207
+ const learned = prepared?.session.metadata.learnedCompatRoles ?? {};
59208
+ const roles = { ...configured, ...learned };
59209
+ const protocol = prepared?.protocol ?? route?.explicitProtocol ?? inferWireProtocol(req.url ?? "");
59210
+ if (protocol === "openai" || protocol === "responses") {
59211
+ compatProtocol = protocol;
59212
+ if (Object.keys(roles).length > 0) {
59213
+ compatRoles = roles;
59214
+ const applied = applyCompatRoles(body, protocol, roles);
59215
+ if (applied.rewritten > 0) {
59216
+ wireBody = applied.body;
59217
+ log2("info", `[${prepared?.session.id ?? "passthrough"}] [compat] rewrote ${applied.rewritten} message role(s) per compat.roles (${Object.entries(roles).map(([f2, t]) => `${f2}\u2192${t}`).join(",")})`);
59218
+ }
59219
+ }
59220
+ }
59221
+ }
59222
+ const wireTransform = compatProtocol ? (b2) => {
59223
+ if (compatRoles) applyCompatRolesJson(b2, compatProtocol, compatRoles);
59224
+ return b2;
59225
+ } : void 0;
58527
59226
  log2("info", `forward ${req.method} \u2192 ${maskUrlForLog(upstreamUrl)}`);
58528
59227
  if (process.env.ACP_DEBUG && prepared) {
58529
59228
  const sid = prepared.session.id;
@@ -58538,9 +59237,9 @@ async function forward(req, res, opts, body, prepared, core, config, log2, route
58538
59237
  }
58539
59238
  }
58540
59239
  }
58541
- if (typeof body === "string" && (opts.debug || bodyDumpEnabled())) {
59240
+ if (typeof wireBody === "string" && (opts.debug || bodyDumpEnabled())) {
58542
59241
  try {
58543
- const parsed = JSON.parse(body);
59242
+ const parsed = JSON.parse(wireBody);
58544
59243
  if (opts.debug) {
58545
59244
  const toolNames = (parsed.tools ?? []).map((t) => {
58546
59245
  const fn = t.function;
@@ -58557,10 +59256,10 @@ async function forward(req, res, opts, body, prepared, core, config, log2, route
58557
59256
  const sid = prepared?.session.id ?? "unknown";
58558
59257
  const out = `${dumpDir}/req-${Date.now()}-${safeSessionId(sid)}.json`;
58559
59258
  try {
58560
- const pretty = JSON.stringify(JSON.parse(body), null, 2);
59259
+ const pretty = JSON.stringify(JSON.parse(wireBody), null, 2);
58561
59260
  fs8.writeFileSync(out, pretty);
58562
59261
  } catch {
58563
- fs8.writeFileSync(out, body);
59262
+ fs8.writeFileSync(out, wireBody);
58564
59263
  }
58565
59264
  log2("info", `[debug] forwarded body written to ${out}`);
58566
59265
  }
@@ -58589,7 +59288,7 @@ async function forward(req, res, opts, body, prepared, core, config, log2, route
58589
59288
  if (rawBase) {
58590
59289
  try {
58591
59290
  const hdrText = Object.entries(maskHeadersForLog(headers)).map(([k2, v2]) => `${k2}: ${v2}`).join("\n");
58592
- const bodyText = req.method === "GET" || req.method === "HEAD" ? "" : typeof body === "string" ? body : Buffer.from(body).toString("utf8");
59291
+ const bodyText = req.method === "GET" || req.method === "HEAD" ? "" : typeof wireBody === "string" ? wireBody : Buffer.from(wireBody).toString("utf8");
58593
59292
  const reqPath = `${rawBase}-REQ.txt`;
58594
59293
  fs8.writeFileSync(reqPath, `${req.method ?? "POST"} ${maskUrlForLog(upstreamUrl)}
58595
59294
  ${hdrText}
@@ -58604,7 +59303,7 @@ ${bodyText}`);
58604
59303
  const init = {
58605
59304
  method: req.method ?? "GET",
58606
59305
  headers,
58607
- body: req.method === "GET" || req.method === "HEAD" ? void 0 : body
59306
+ body: req.method === "GET" || req.method === "HEAD" ? void 0 : wireBody
58608
59307
  };
58609
59308
  if (dispatcher) init.dispatcher = dispatcher;
58610
59309
  const clientAbort = new AbortController();
@@ -58619,6 +59318,49 @@ ${bodyText}`);
58619
59318
  recordUpstreamConnection(upstreamUrl, proxyUrl, error);
58620
59319
  throw new Error(`upstream request failed: ${formatUpstreamError(error, upstreamUrl, proxyUrl)}`, { cause: error });
58621
59320
  }
59321
+ if (compatProtocol && typeof wireBody === "string" && upstreamResult.response.status === 400 && upstreamResult.response.body) {
59322
+ let roleErrText = null;
59323
+ try {
59324
+ roleErrText = (await readStreamToBuffer(upstreamResult.response.body)).toString("utf8");
59325
+ } catch {
59326
+ roleErrText = null;
59327
+ }
59328
+ if (roleErrText !== null) {
59329
+ upstreamResult = {
59330
+ response: new Response(roleErrText, {
59331
+ status: upstreamResult.response.status,
59332
+ statusText: upstreamResult.response.statusText,
59333
+ headers: new Headers(upstreamResult.response.headers)
59334
+ }),
59335
+ clearTimer: upstreamResult.clearTimer
59336
+ };
59337
+ const rejection = detectRoleRejection(upstreamResult.response.status, roleErrText);
59338
+ if (rejection && rejection.role !== "system") {
59339
+ const fixed = applyCompatRoles(wireBody, compatProtocol, { [rejection.role]: "system" });
59340
+ if (fixed.rewritten > 0) {
59341
+ try {
59342
+ const retry = await fetchWithTimeout(upstreamUrl, { ...init, body: fixed.body }, void 0, clientAbort.signal);
59343
+ if (retry.response.ok) {
59344
+ upstreamResult.clearTimer();
59345
+ const s3 = prepared?.session;
59346
+ if (s3) {
59347
+ const prev = s3.metadata.learnedCompatRoles ?? {};
59348
+ s3.metadata.learnedCompatRoles = { ...prev, [rejection.role]: "system" };
59349
+ markDirty(s3);
59350
+ }
59351
+ compatRoles = { ...compatRoles, [rejection.role]: "system" };
59352
+ const providerKey = new URL(upstreamUrl).origin;
59353
+ log2("info", `[${prepared?.session.id ?? "passthrough"}] [compat] upstream rejected role "${rejection.role}" \u2014 auto-rewrote ${fixed.rewritten} message role(s) to "system", retry OK (remembered for this session only). To make permanent, add: {"providers":{"${providerKey}":{"compat":{"roles":{"${rejection.role}":"system"}}}}`);
59354
+ upstreamResult = retry;
59355
+ } else {
59356
+ retry.clearTimer();
59357
+ }
59358
+ } catch {
59359
+ }
59360
+ }
59361
+ }
59362
+ }
59363
+ }
58622
59364
  const { response: upstream, clearTimer: clearUpstreamTimer } = upstreamResult;
58623
59365
  const respHeaders = {};
58624
59366
  const respConnNamed = connectionNamedHeaders(upstream.headers.get("connection") ?? void 0);
@@ -58664,9 +59406,10 @@ ${hdrText}
58664
59406
  if (info.isOverflow) {
58665
59407
  let reqModel;
58666
59408
  let rejectedImageTokens = 0;
59409
+ let parsedBody;
58667
59410
  try {
58668
59411
  const rawBody = typeof prepared.body === "string" ? prepared.body : prepared.body.toString("utf8");
58669
- const parsedBody = JSON.parse(rawBody);
59412
+ parsedBody = JSON.parse(rawBody);
58670
59413
  reqModel = typeof parsedBody.model === "string" ? parsedBody.model : void 0;
58671
59414
  rejectedImageTokens = imageTokensInParsedBody(prepared.protocol, parsedBody);
58672
59415
  } catch {
@@ -58680,7 +59423,7 @@ ${hdrText}
58680
59423
  s3.metadata.learnedContextLimits = learnedMap;
58681
59424
  log2("warn", `[${s3.id}] upstream context overflow \u2014 learned real window ${info.window} for ${reqModel ?? "(unknown model)"} (was ${prev ?? "unset"}); arming emergency shrink`);
58682
59425
  } else {
58683
- const payloadEstimate = estimateCoreMessages(prepared.processedMessages) + rejectedImageTokens;
59426
+ const payloadEstimate = (prepared.processedMessages.length > 0 ? estimateCoreMessages(prepared.processedMessages) : estimateRawBodyTokens(parsedBody)) + rejectedImageTokens;
58684
59427
  const prev = (reqModel ? learnedMap[reqModel] : void 0) ?? s3.metadata.learnedContextLimit;
58685
59428
  if (payloadEstimate >= 1e3 && (prev === void 0 || payloadEstimate < prev)) {
58686
59429
  if (reqModel) learnedMap[reqModel] = payloadEstimate;
@@ -58704,6 +59447,18 @@ ${hdrText}
58704
59447
  if (bodyText.length > 600) snippet += " \u2026";
58705
59448
  if (!snippet) snippet = "(no body)";
58706
59449
  log("warn", `[${errSid}] \u2190 upstream ${upstream.status}${reqIdText}: ${snippet}`);
59450
+ if (res.headersSent) {
59451
+ if (prepared?.stream) {
59452
+ emitStreamError(res, prepared.protocol, `upstream HTTP ${upstream.status}: ${snippet}`, (m2) => log("info", m2));
59453
+ } else {
59454
+ try {
59455
+ res.end(errBody ?? void 0);
59456
+ } catch {
59457
+ }
59458
+ }
59459
+ clearUpstreamTimer();
59460
+ return;
59461
+ }
58707
59462
  const errHeaders = { ...respHeaders };
58708
59463
  delete errHeaders["content-length"];
58709
59464
  delete errHeaders["transfer-encoding"];
@@ -58712,7 +59467,7 @@ ${hdrText}
58712
59467
  clearUpstreamTimer();
58713
59468
  return;
58714
59469
  }
58715
- res.writeHead(upstream.status, respHeaders);
59470
+ if (!res.headersSent) res.writeHead(upstream.status, respHeaders);
58716
59471
  if (!upstream.body) {
58717
59472
  res.end();
58718
59473
  clearUpstreamTimer();
@@ -58820,7 +59575,7 @@ ${hdrText}
58820
59575
  const reqHeaders = buildForwardHeaders(headers);
58821
59576
  const textProtocol = prepared.protocol === "responses" && !!prepared.responsesTextProtocol;
58822
59577
  const systemPrompt = textProtocol ? buildCompressHybridSystemPrompt(prepared.prompts ?? defaultPrompts) : buildCompressSystemPrompt(prepared.prompts ?? defaultPrompts);
58823
- const adapter = pickAdapter(prepared.protocol, parsedReq, textProtocol, prepared.responsesProjection, prepared.anthropicSystem, prepared.openaiSystemText);
59578
+ const adapter = pickAdapter(prepared.protocol, parsedReq, textProtocol, prepared.responsesProjection, prepared.anthropicSystem, prepared.openaiSystemText, prepared.session.hostCreditTokens ?? 0);
58824
59579
  const refreshFolded = (current) => {
58825
59580
  const turn = core.processTurn({
58826
59581
  messages: prepared.originalMessages,
@@ -58831,13 +59586,13 @@ ${hdrText}
58831
59586
  });
58832
59587
  prepared.session.state = turn.state;
58833
59588
  const records = current.filter((m2) => typeof m2.id === "string" && m2.id.startsWith("acp_loop_"));
58834
- return stripKernelSummaries([...turn.messages, ...records], turn.state);
59589
+ return repairResponsesAssistantOrdering(stripKernelSummaries([...turn.messages, ...records], turn.state), prepared.originalMessages);
58835
59590
  };
58836
59591
  const loop = runCompressLoop(
58837
59592
  streamToRead,
58838
- { core, config, messages: prepared.processedMessages.length > 0 ? prepared.processedMessages : prepared.originalMessages, compressMessages: prepared.originalMessages, session: prepared.session, log: ctx.log, proxyUrl, protocol: prepared.protocol, textProtocol, debug: opts.debug, nudge: prepared.nudge, refreshFolded },
59593
+ { core, config, messages: prepared.processedMessages.length > 0 ? prepared.processedMessages : prepared.originalMessages, compressMessages: prepared.originalMessages, session: prepared.session, log: ctx.log, proxyUrl, protocol: prepared.protocol, textProtocol, debug: opts.debug, refreshFolded },
58839
59594
  parsedReq,
58840
- { url: upstreamUrl, headers: reqHeaders },
59595
+ { url: upstreamUrl, headers: reqHeaders, wireTransform },
58841
59596
  adapter,
58842
59597
  systemPrompt,
58843
59598
  clientAbort.signal
@@ -58884,7 +59639,7 @@ ${hdrText}
58884
59639
  json,
58885
59640
  { core, config, messages: prepared.originalMessages, session: prepared.session, log: ctx.log, proxyUrl, textProtocol: true },
58886
59641
  requestBody,
58887
- { url: upstreamUrl, headers: requestHeaders }
59642
+ { url: upstreamUrl, headers: requestHeaders, wireTransform }
58888
59643
  );
58889
59644
  }
58890
59645
  const u2 = json.usage ?? {};
@@ -58902,6 +59657,10 @@ ${hdrText}
58902
59657
  const out = u2.completion_tokens ?? u2.output_tokens;
58903
59658
  if (typeof out === "number") prepared.session.stats.outputTokens += out;
58904
59659
  }
59660
+ const credit = prepared.session.hostCreditTokens ?? 0;
59661
+ if (credit > 0 && backfillHostUsage(prepared.protocol, u2, credit)) {
59662
+ prepared.session.hostContextTokens = (typeof total === "number" ? total : 0) + credit;
59663
+ }
58905
59664
  if (prepared.protocol === "openai") {
58906
59665
  rewriteOpenaiJsonResponse(json, ctx);
58907
59666
  } else if (prepared.protocol === "responses") {
@@ -59045,6 +59804,7 @@ function handleConfigReload(opts, res, log2) {
59045
59804
  for (const k2 of Object.keys(opts.routes)) delete opts.routes[k2];
59046
59805
  Object.assign(opts.routes, fresh);
59047
59806
  opts.compress = loadOptions().compress;
59807
+ opts.compat = loadOptions().compat;
59048
59808
  resetProxyCache();
59049
59809
  const names = Object.keys(fresh);
59050
59810
  log2("info", `[acp-web] routes hot-reloaded (${names.length} providers): ${names.join(", ") || "(none)"}`);
@@ -62148,6 +62908,14 @@ async function findInstallDir(packageName) {
62148
62908
  dir = parent;
62149
62909
  }
62150
62910
  }
62911
+ async function isGitWorkingTree(dir) {
62912
+ try {
62913
+ await access(path12.join(dir, ".git"));
62914
+ return true;
62915
+ } catch {
62916
+ return false;
62917
+ }
62918
+ }
62151
62919
  async function readDiskVersion(installDir) {
62152
62920
  try {
62153
62921
  const pkg = JSON.parse(await readFile2(path12.join(installDir, "package.json"), "utf-8"));
@@ -62304,6 +63072,11 @@ async function checkForUpdate(opts, force = false) {
62304
63072
  }
62305
63073
  await writeLastCheck(now);
62306
63074
  firstCheckDone = true;
63075
+ const installDir = await findInstallDir(opts.packageName);
63076
+ if (installDir && await isGitWorkingTree(installDir)) {
63077
+ log("info", `[update] running from a source checkout (${installDir}) \u2014 skipping auto-update (use npm install -g ${opts.packageName})`);
63078
+ return;
63079
+ }
62307
63080
  log("info", `[update] checking npm registry for ${opts.packageName}${sinceLastSec < 0 ? " (startup check)" : sinceLastSec === 0 ? "" : ` (last check ${sinceLastSec}s ago)`}\u2026`);
62308
63081
  const url = `${REGISTRY_BASE}/${opts.packageName}/latest`;
62309
63082
  const res = await fetch(url, {
@@ -62320,7 +63093,6 @@ async function checkForUpdate(opts, force = false) {
62320
63093
  log("warn", `[update] registry response had no version, skipping`);
62321
63094
  return;
62322
63095
  }
62323
- const installDir = await findInstallDir(opts.packageName);
62324
63096
  const diskVersion = installDir ? await readDiskVersion(installDir) : void 0;
62325
63097
  const currentVersion = diskVersion ?? opts.currentVersion;
62326
63098
  if (!isNewer(latest, currentVersion)) {
@@ -62391,6 +63163,9 @@ async function installViaTarball(version2, tarballUrl, installDir, integrity, sh
62391
63163
  } catch {
62392
63164
  return { ok: false, error: `install dir not writable: ${installDir}` };
62393
63165
  }
63166
+ if (await isGitWorkingTree(installDir)) {
63167
+ return { ok: false, error: `install dir is a git working tree (${installDir}) \u2014 refusing to overwrite a source checkout (use npm install -g)` };
63168
+ }
62394
63169
  const MAX_TARBALL_BYTES = 100 * 1024 * 1024;
62395
63170
  let tgzBuffer;
62396
63171
  try {
@@ -63104,6 +63879,11 @@ function unwrapUpstream2(url) {
63104
63879
  const idx = url.indexOf("/bili/");
63105
63880
  return idx >= 0 ? url.slice(idx + "/bili/".length) : url;
63106
63881
  }
63882
+ function isLoopbackHost(host) {
63883
+ const h = host.toLowerCase();
63884
+ if (h === "localhost" || h === "::1" || h === "[::1]") return true;
63885
+ return /^127\.\d+\.\d+\.\d+$/.test(h);
63886
+ }
63107
63887
  function resolveCaCertPath(env) {
63108
63888
  const base = env.XDG_DATA_HOME || path15.join(os5.homedir(), ".local/share");
63109
63889
  return path15.join(base, "billion-context", "ca", "root-ca.pem");
@@ -63116,6 +63896,7 @@ function discoverRoutes(client, config) {
63116
63896
  const httpsDomains = [];
63117
63897
  const httpRewrites = [];
63118
63898
  const httpsRewrites = [];
63899
+ const httpEnvRoutes = [];
63119
63900
  const httpsSeen = /* @__PURE__ */ new Set();
63120
63901
  const rewriteKeys = /* @__PURE__ */ new Set();
63121
63902
  const httpsRewriteKeys = /* @__PURE__ */ new Set();
@@ -63201,8 +63982,18 @@ function discoverRoutes(client, config) {
63201
63982
  if (dshSeen.has(real)) continue;
63202
63983
  dshSeen.add(real);
63203
63984
  anon += 1;
63204
- rewriteKeys.add(`dsh-${anon}`);
63205
- httpRewrites.push({ key: `dsh-${anon}`, realUpstream: real });
63985
+ if (isLoopbackHost(url.hostname)) {
63986
+ rewriteKeys.add(`dsh-${anon}`);
63987
+ httpRewrites.push({ key: `dsh-${anon}`, realUpstream: real });
63988
+ } else if (url.protocol === "https:") {
63989
+ const host = url.hostname.toLowerCase();
63990
+ if (host && !httpsSeen.has(host)) {
63991
+ httpsSeen.add(host);
63992
+ httpsDomains.push(host);
63993
+ }
63994
+ } else if (!httpEnvRoutes.includes(real)) {
63995
+ httpEnvRoutes.push(real);
63996
+ }
63206
63997
  } catch {
63207
63998
  }
63208
63999
  }
@@ -63212,7 +64003,7 @@ function discoverRoutes(client, config) {
63212
64003
  }
63213
64004
  classify(config.codex?.openaiBaseUrl, "openai_base_url");
63214
64005
  }
63215
- return { httpsDomains, httpRewrites, httpsRewrites };
64006
+ return { httpsDomains, httpRewrites, httpsRewrites, httpEnvRoutes };
63216
64007
  }
63217
64008
  function discoverDomains(client, config) {
63218
64009
  return discoverRoutes(client, config).httpsDomains;
@@ -64116,7 +64907,7 @@ async function runLaunch(params, deps = {}) {
64116
64907
  const domains = dedupeInOrder([...routes.httpsDomains, ...params.mitmDomains ?? []]);
64117
64908
  const handle2 = await ensureProxyRunning({ host, port, passthrough, debug, mitmDomains: domains, modelWindows: collectModelWindows(config, base) }, deps);
64118
64909
  console.error(
64119
- `bili: started proxy at ${handle2.origin} (MITM domains: ${domains.length ? domains.join(", ") : "defaults"})` + (routes.httpRewrites.length > 0 ? ` (HTTP /bili/ rewrites: ${routes.httpRewrites.length})` : "") + (routes.httpsRewrites.length > 0 ? ` (HTTPS cert rewrites: ${routes.httpsRewrites.length})` : "") + (params.client === "pi-test" ? " (no extensions)" : "")
64910
+ `bili: started proxy at ${handle2.origin} (MITM domains: ${domains.length ? domains.join(", ") : "defaults"})` + (routes.httpRewrites.length > 0 ? ` (HTTP /bili/ rewrites: ${routes.httpRewrites.length})` : "") + (routes.httpsRewrites.length > 0 ? ` (HTTPS cert rewrites: ${routes.httpsRewrites.length})` : "") + (routes.httpEnvRoutes.length > 0 ? ` (HTTP proxy-env routes: ${routes.httpEnvRoutes.length})` : "") + (params.client === "pi-test" ? " (no extensions)" : "")
64120
64911
  );
64121
64912
  if (handle2.logPath) {
64122
64913
  console.error(`bili: proxy log: ${handle2.logPath}`);
@@ -64181,18 +64972,25 @@ async function runLaunch(params, deps = {}) {
64181
64972
  );
64182
64973
  }
64183
64974
  } else if (base === "dsh") {
64184
- env = { ...process.env, BILLION_CONTEXT_PROXY: origin };
64975
+ const usesProxyEnv = routes.httpsDomains.length > 0 || routes.httpEnvRoutes.length > 0;
64976
+ env = usesProxyEnv ? stripInheritedProxy(process.env) : { ...process.env };
64977
+ env.BILLION_CONTEXT_PROXY = origin;
64185
64978
  env.DEEPSEEK_BASE_URL = wrapUpstream(origin, "https://api.deepseek.com");
64979
+ if (usesProxyEnv) {
64980
+ env.HTTPS_PROXY = origin;
64981
+ env.SSL_CERT_FILE = resolveCombinedCaPath(process.env);
64982
+ }
64983
+ if (routes.httpEnvRoutes.length > 0) env.HTTP_PROXY = origin;
64186
64984
  env.PI_CACHE_RETENTION = "long";
64187
64985
  const dshHomeDir = resolveDshHome(process.env);
64188
- dshOverlayHome = prepareDshHome(dshHomeDir, origin, routes.httpRewrites);
64986
+ dshOverlayHome = routes.httpRewrites.length > 0 ? prepareDshHome(dshHomeDir, origin, routes.httpRewrites) : void 0;
64189
64987
  if (dshOverlayHome) {
64190
64988
  env.DSH_HOME = dshOverlayHome;
64191
64989
  } else if (routes.httpRewrites.length > 0) {
64192
64990
  console.error(
64193
- "bili: dsh settings.yaml could not be rewritten (unreadable or no matching endpoints) \u2014 custom providers will NOT go through the proxy; the built-in deepseek route still does."
64991
+ "bili: dsh settings.yaml could not be rewritten (unreadable or no matching endpoints) \u2014 loopback custom providers will NOT go through the proxy; other routes still do."
64194
64992
  );
64195
- } else {
64993
+ } else if (!usesProxyEnv) {
64196
64994
  console.error(
64197
64995
  "bili: no custom providers found in ~/.dsh/settings.yaml \u2014 proxying the built-in deepseek route via DEEPSEEK_BASE_URL only."
64198
64996
  );
@@ -64355,10 +65153,13 @@ function renderHandoff(s3, full) {
64355
65153
  lines.push("");
64356
65154
  const messages = s3.lastMessages;
64357
65155
  if (messages && messages.length > 0) {
64358
- lines.push(full ? `## Full conversation (${messages.length} messages)` : `## Conversation (folded view as the model saw it, ${messages.length} client messages)`);
65156
+ const folded = s3.lastMessagesFolded === true;
65157
+ const view = full || folded ? messages : prune(messages, s3.state);
65158
+ lines.push(
65159
+ full && !folded ? `## Full conversation (${messages.length} messages)` : folded ? `## Conversation (persisted folded snapshot, ${messages.length} messages)` : `## Conversation (folded view as the model saw it, ${messages.length} client messages)`
65160
+ );
64359
65161
  lines.push("");
64360
65162
  let lastRole = "";
64361
- const view = full ? messages : prune(messages, s3.state);
64362
65163
  for (const m2 of view) {
64363
65164
  if (m2.role !== lastRole) {
64364
65165
  lines.push(`### ${m2.role}`);
@@ -64368,6 +65169,18 @@ function renderHandoff(s3, full) {
64368
65169
  lines.push(renderMessage2(m2));
64369
65170
  }
64370
65171
  lines.push("");
65172
+ if (full && folded) {
65173
+ for (const b2 of s3.state.blocks.filter((x) => x.active)) {
65174
+ const content = s3.blockContents.get(b2.blockId);
65175
+ if (!content) continue;
65176
+ lines.push(`## Block ${b2.blockId}${b2.topic ? ` \u2014 ${b2.topic}` : ""}`);
65177
+ lines.push("");
65178
+ lines.push(`### Original messages (${content.full.count})`);
65179
+ lines.push("");
65180
+ lines.push(content.full.text.trim());
65181
+ lines.push("");
65182
+ }
65183
+ }
64371
65184
  return lines.join("\n");
64372
65185
  }
64373
65186
  const active = s3.state.blocks.filter((b2) => b2.active);
@@ -64429,7 +65242,7 @@ async function exportSession(selector, opts = {}) {
64429
65242
  const store = new SessionStore({ dir: opts.dir, enabled: true });
64430
65243
  const all = [...(await store.loadAll()).values()];
64431
65244
  if (all.length === 0) {
64432
- return "No persisted sessions found. Sessions are written under the sessions directory once the proxy has served a request (compression state and compressed originals only \u2014 uncompressed conversation text is not persisted).";
65245
+ return "No persisted sessions found. Sessions are written under the sessions directory once the proxy has served a request (compression state, compressed originals, and a bounded folded-view snapshot of the recent conversation).";
64433
65246
  }
64434
65247
  if (!selector) {
64435
65248
  const list = await listSessions2(opts);