billion-context 0.1.92-pr.622.446 → 0.1.92

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -5632,7 +5632,7 @@ var require_util2 = __commonJS({
5632
5632
  callback();
5633
5633
  }
5634
5634
  };
5635
- function createInflate2(zlibOptions) {
5635
+ function createInflate(zlibOptions) {
5636
5636
  return new InflateStream(zlibOptions);
5637
5637
  }
5638
5638
  function extractMimeType(headers) {
@@ -5764,7 +5764,7 @@ var require_util2 = __commonJS({
5764
5764
  readAllBytes,
5765
5765
  simpleRangeHeaderValue,
5766
5766
  buildContentRange,
5767
- createInflate: createInflate2,
5767
+ createInflate,
5768
5768
  extractMimeType,
5769
5769
  getDecodeSplit,
5770
5770
  environmentSettingsObject,
@@ -13122,7 +13122,7 @@ var require_mock_interceptor = __commonJS({
13122
13122
  var require_mock_client = __commonJS({
13123
13123
  "node_modules/undici/lib/mock/mock-client.js"(exports, module) {
13124
13124
  "use strict";
13125
- var { promisify } = __require("util");
13125
+ var { promisify: promisify2 } = __require("util");
13126
13126
  var Client = require_client();
13127
13127
  var { buildMockDispatch } = require_mock_utils();
13128
13128
  var {
@@ -13170,7 +13170,7 @@ var require_mock_client = __commonJS({
13170
13170
  this[kDispatches] = [];
13171
13171
  }
13172
13172
  async [kClose]() {
13173
- await promisify(this[kOriginalClose])();
13173
+ await promisify2(this[kOriginalClose])();
13174
13174
  this[kConnected] = 0;
13175
13175
  this[kMockAgent][Symbols.kClients].delete(this[kOrigin]);
13176
13176
  }
@@ -13383,7 +13383,7 @@ var require_mock_call_history = __commonJS({
13383
13383
  var require_mock_pool = __commonJS({
13384
13384
  "node_modules/undici/lib/mock/mock-pool.js"(exports, module) {
13385
13385
  "use strict";
13386
- var { promisify } = __require("util");
13386
+ var { promisify: promisify2 } = __require("util");
13387
13387
  var Pool2 = require_pool();
13388
13388
  var { buildMockDispatch } = require_mock_utils();
13389
13389
  var {
@@ -13431,7 +13431,7 @@ var require_mock_pool = __commonJS({
13431
13431
  this[kDispatches] = [];
13432
13432
  }
13433
13433
  async [kClose]() {
13434
- await promisify(this[kOriginalClose])();
13434
+ await promisify2(this[kOriginalClose])();
13435
13435
  this[kConnected] = 0;
13436
13436
  this[kMockAgent][Symbols.kClients].delete(this[kOrigin]);
13437
13437
  }
@@ -17545,17 +17545,17 @@ var require_cache2 = __commonJS({
17545
17545
  var require_decompress = __commonJS({
17546
17546
  "node_modules/undici/lib/interceptor/decompress.js"(exports, module) {
17547
17547
  "use strict";
17548
- var { createInflate: createInflate2, createGunzip: createGunzip2, createBrotliDecompress: createBrotliDecompress2, createZstdDecompress } = __require("zlib");
17548
+ var { createInflate, createGunzip, createBrotliDecompress, createZstdDecompress } = __require("zlib");
17549
17549
  var { pipeline } = __require("stream");
17550
17550
  var DecoratorHandler = require_decorator_handler();
17551
17551
  var { runtimeFeatures } = require_runtime_features();
17552
17552
  var supportedEncodings = {
17553
- gzip: createGunzip2,
17554
- "x-gzip": createGunzip2,
17555
- br: createBrotliDecompress2,
17556
- deflate: createInflate2,
17557
- compress: createInflate2,
17558
- "x-compress": createInflate2,
17553
+ gzip: createGunzip,
17554
+ "x-gzip": createGunzip,
17555
+ br: createBrotliDecompress,
17556
+ deflate: createInflate,
17557
+ compress: createInflate,
17558
+ "x-compress": createInflate,
17559
17559
  ...runtimeFeatures.has("zstd") ? { zstd: createZstdDecompress } : {}
17560
17560
  };
17561
17561
  var defaultSkipStatusCodes = (
@@ -20384,7 +20384,7 @@ var require_fetch = __commonJS({
20384
20384
  clampAndCoarsenConnectionTimingInfo,
20385
20385
  simpleRangeHeaderValue,
20386
20386
  buildContentRange,
20387
- createInflate: createInflate2,
20387
+ createInflate,
20388
20388
  extractMimeType,
20389
20389
  hasAuthenticationEntry,
20390
20390
  includesCredentials,
@@ -21366,7 +21366,7 @@ var require_fetch = __commonJS({
21366
21366
  finishFlush: zlib.constants.Z_SYNC_FLUSH
21367
21367
  }));
21368
21368
  } else if (coding === "deflate") {
21369
- decoders.push(createInflate2({
21369
+ decoders.push(createInflate({
21370
21370
  flush: zlib.constants.Z_SYNC_FLUSH,
21371
21371
  finishFlush: zlib.constants.Z_SYNC_FLUSH
21372
21372
  }));
@@ -23432,7 +23432,7 @@ var require_connection = __commonJS({
23432
23432
  var require_permessage_deflate = __commonJS({
23433
23433
  "node_modules/undici/lib/web/websocket/permessage-deflate.js"(exports, module) {
23434
23434
  "use strict";
23435
- var { createInflateRaw: createInflateRaw2, Z_DEFAULT_WINDOWBITS } = __require("zlib");
23435
+ var { createInflateRaw, Z_DEFAULT_WINDOWBITS } = __require("zlib");
23436
23436
  var { isValidClientWindowBits } = require_util5();
23437
23437
  var { MessageSizeExceededError } = require_errors();
23438
23438
  var tail = Buffer.from([0, 0, 255, 255]);
@@ -23468,7 +23468,7 @@ var require_permessage_deflate = __commonJS({
23468
23468
  windowBits = Number.parseInt(this.#options.serverMaxWindowBits);
23469
23469
  }
23470
23470
  try {
23471
- this.#inflate = createInflateRaw2({ windowBits });
23471
+ this.#inflate = createInflateRaw({ windowBits });
23472
23472
  } catch (err2) {
23473
23473
  callback(err2);
23474
23474
  return;
@@ -38339,7 +38339,7 @@ var require_tls = __commonJS({
38339
38339
  }
38340
38340
  return rval;
38341
38341
  };
38342
- var inflate = function(c, record, s3) {
38342
+ var inflate2 = function(c, record, s3) {
38343
38343
  var rval = false;
38344
38344
  try {
38345
38345
  var bytes = c.inflate(record.fragment.getBytes());
@@ -39484,7 +39484,7 @@ var require_tls = __commonJS({
39484
39484
  case tls4.CompressionMethod.none:
39485
39485
  break;
39486
39486
  case tls4.CompressionMethod.deflate:
39487
- state.read.compressFunction = inflate;
39487
+ state.read.compressFunction = inflate2;
39488
39488
  state.write.compressFunction = deflate;
39489
39489
  break;
39490
39490
  default:
@@ -43476,6 +43476,10 @@ function defaultCountTokens(text) {
43476
43476
  const cjkCount = cjk?.length ?? 0;
43477
43477
  return cjkCount + Math.ceil((text.length - cjkCount) / 4);
43478
43478
  }
43479
+ function estimateTokensFast(text) {
43480
+ if (!text) return 0;
43481
+ return Math.ceil(text.length / 4);
43482
+ }
43479
43483
  var COMPRESS_PHILOSOPHY = `Compression Philosophy:
43480
43484
  - All compression serves the primary task, but be frugal.
43481
43485
  - Context capacity is precious. Save context by compressing consumed outputs, not by avoiding tools.
@@ -44917,6 +44921,11 @@ function formatTokenCount(tokens) {
44917
44921
  function buildAbsorbPrompt(ref, tokens, toolName = ABSORB_TOOL_NAME) {
44918
44922
  return `${ABSORB_PROMPT_MARKER} This tool result (~${formatTokenCount(tokens)} tokens) will be REMOVED from context. Your IMMEDIATE next action: call ${toolName}({ ref: "${ref}", summary: "..." }) \u2014 summary = distilled essentials only (outcome, key values, exact paths:lines, error text verbatim, decisions). Afterwards work from your summary; do NOT re-run this tool. If the result contains nothing you need, call ${toolName} with summary "(nothing needed)".`;
44919
44923
  }
44924
+ function buildAbsorbSystemPrompt(toolName = ABSORB_TOOL_NAME) {
44925
+ return `INSTANT TOOL-RESULT ABSORPTION (${toolName})
44926
+
44927
+ Some tool results end with a ${ABSORB_PROMPT_MARKER} instruction. When you see one, your IMMEDIATE next action must be calling ${toolName}({ ref, summary }) \u2014 distill that tool result's essentials into summary: outcome, key values, exact paths:lines, error text verbatim, decisions. The original output is then removed from context; your ${toolName} summary becomes the only durable record of it, so distill carefully. Never call another tool or answer the user before absorbing a marked result. Do not re-run the original tool afterwards \u2014 work from your summary. ${toolName} calls are ordinary context: the regular compression system may fold them later like any other message.`;
44928
+ }
44920
44929
  function isAcpOrConfiguredTool(toolName, cfg) {
44921
44930
  if (!toolName) return false;
44922
44931
  if (toolName === cfg.toolName) return true;
@@ -48104,6 +48113,19 @@ function detectRoleRejection(status, bodyText) {
48104
48113
  }
48105
48114
  return null;
48106
48115
  }
48116
+ function detectSystemPlacementError(status, bodyText) {
48117
+ if (status !== 400) return false;
48118
+ const head = bodyText.slice(0, 4096);
48119
+ if (!/system/i.test(head)) return false;
48120
+ const norm = head.replace(/['"`[\](){}]/g, " ");
48121
+ const PLACEMENT_MARKERS = [
48122
+ /\b(multiple|more\s+than\s+one|exactly\s+one|only\s+one|single|another|second|third)\s+system/i,
48123
+ /\b[2-9]\d*\s+system\s+messages?\b/i,
48124
+ /system\s+messages?\b[^.\n]{0,50}\bindex\s*\d+\b/i,
48125
+ /(system\s+messages?|system\s+roles?)\b[^.\n]{0,50}\b(beginning|start|front|first\s+message|must\s+be\s+first)/i
48126
+ ];
48127
+ return PLACEMENT_MARKERS.some((re2) => re2.test(norm));
48128
+ }
48107
48129
  function applyCompatRoles(body, protocol, roles) {
48108
48130
  if (Object.keys(roles).length === 0) return { body, rewritten: 0 };
48109
48131
  let parsed;
@@ -48575,6 +48597,69 @@ function resolveRequestConfig(base, routes, embeddedUrl, model, native, globalCo
48575
48597
  return applyCompressSettings(base, limit, compress);
48576
48598
  }
48577
48599
 
48600
+ // src/strip-images.ts
48601
+ var DEFAULT_STRIP_IMAGES_KEEP_RECENT = 5;
48602
+ var IMAGE_PLACEHOLDER = "[image]";
48603
+ function isObj(v2) {
48604
+ return typeof v2 === "object" && v2 !== null;
48605
+ }
48606
+ function isImagePart(protocol, part) {
48607
+ if (!isObj(part)) return false;
48608
+ if (protocol === "responses") return part.type === "input_image";
48609
+ if (protocol === "openai") return part.type === "image_url";
48610
+ return part.type === "image";
48611
+ }
48612
+ function placeholderContent(protocol) {
48613
+ const type = protocol === "responses" ? "input_text" : "text";
48614
+ return [{ type, text: IMAGE_PLACEHOLDER }];
48615
+ }
48616
+ function stripHistoricalImages(body, protocol, keepRecent) {
48617
+ if (!protocol || !isObj(body)) return { body, removed: 0 };
48618
+ const recentCount = Math.max(0, Math.floor(keepRecent));
48619
+ if (protocol === "responses") {
48620
+ const input = body.input;
48621
+ if (!Array.isArray(input)) return { body, removed: 0 };
48622
+ const cutoff2 = input.length - recentCount;
48623
+ let removed2 = 0;
48624
+ let touched2 = false;
48625
+ const nextInput = input.map((item, i) => {
48626
+ if (i < cutoff2 && isObj(item) && Array.isArray(item.content)) {
48627
+ const content = item.content;
48628
+ const imgs = content.filter((p2) => isImagePart("responses", p2)).length;
48629
+ if (imgs > 0) {
48630
+ removed2 += imgs;
48631
+ touched2 = true;
48632
+ const kept = content.filter((p2) => !isImagePart("responses", p2));
48633
+ return { ...item, content: kept.length > 0 ? kept : placeholderContent("responses") };
48634
+ }
48635
+ }
48636
+ return item;
48637
+ });
48638
+ if (!touched2) return { body, removed: 0 };
48639
+ return { body: { ...body, input: nextInput }, removed: removed2 };
48640
+ }
48641
+ const messages = body.messages;
48642
+ if (!Array.isArray(messages)) return { body, removed: 0 };
48643
+ const cutoff = messages.length - recentCount;
48644
+ let removed = 0;
48645
+ let touched = false;
48646
+ const nextMessages = messages.map((m2, i) => {
48647
+ if (i < cutoff && isObj(m2) && Array.isArray(m2.content)) {
48648
+ const content = m2.content;
48649
+ const imgs = content.filter((p2) => isImagePart(protocol, p2)).length;
48650
+ if (imgs > 0) {
48651
+ removed += imgs;
48652
+ touched = true;
48653
+ const kept = content.filter((p2) => !isImagePart(protocol, p2));
48654
+ return { ...m2, content: kept.length > 0 ? kept : placeholderContent(protocol) };
48655
+ }
48656
+ }
48657
+ return m2;
48658
+ });
48659
+ if (!touched) return { body, removed: 0 };
48660
+ return { body: { ...body, messages: nextMessages }, removed };
48661
+ }
48662
+
48578
48663
  // src/registry.ts
48579
48664
  import { readFile, writeFile, mkdir } from "fs/promises";
48580
48665
  import { existsSync as existsSync2, statSync as statSync2 } from "fs";
@@ -50964,52 +51049,14 @@ function withStagedCompressGuidance(text) {
50964
51049
  return text + STAGED_COMPRESS_GUIDANCE;
50965
51050
  }
50966
51051
 
50967
- // src/acp-status.ts
50968
- function handleAcpStatus(args, ctx) {
50969
- const scope = typeof args.scope === "string" ? args.scope : void 0;
50970
- const view = typeof args.view === "string" ? args.view : void 0;
50971
- const tool = typeof args.tool === "string" ? args.tool : void 0;
50972
- const sort = typeof args.sort === "string" ? args.sort : void 0;
50973
- const limit = typeof args.limit === "number" ? args.limit : void 0;
50974
- const base = buildStatusReport(ctx.session.state, ctx.messages, defaultCountTokens, { scope, view, tool, sort, limit });
50975
- if (scope) return base;
50976
- const extra = [];
50977
- try {
50978
- const turn = ctx.core.processTurn({
50979
- messages: ctx.messages,
50980
- state: ctx.session.state,
50981
- config: ctx.config,
50982
- tokenCount: ctx.session.stats.lastInputTokens,
50983
- renderTags: "none"
50984
- });
50985
- const nudge = turn.nudge;
50986
- if (nudge) {
50987
- extra.push("");
50988
- extra.push(nudge.shouldInject ? `Nudge: ACTIVE \u2014 ${nudge.reason}` : `Nudge: idle \u2014 ${nudge.reason}`);
50989
- const ranges = viableRanges(nudge.compressibleRanges);
50990
- const protectedRanges = nudge.protectedRanges ?? [];
50991
- if (ranges.length > 0 || protectedRanges.length > 0) {
50992
- extra.push("");
50993
- extra.push(formatRanges(ranges, protectedRanges));
50994
- }
50995
- }
50996
- } catch {
50997
- }
50998
- const archive = preCompactionArchiveOf(ctx.session);
50999
- const archivedIds = Object.keys(archive);
51000
- if (archivedIds.length > 0) {
51001
- extra.push("");
51002
- extra.push(`PRE-COMPACTION ARCHIVE \u2014 ${archivedIds.length} block(s): content was replaced by the client's native compaction summary, so it is no longer in the session history and decompress is unavailable.`);
51003
- for (const id of archivedIds) {
51004
- extra.push(` ${id} \u2014 ${archive[id].reason}`);
51005
- }
51006
- }
51007
- return extra.length > 0 ? `${base}
51008
- ${extra.join("\n")}` : base;
51009
- }
51010
-
51011
51052
  // src/absorb.ts
51012
51053
  var EFFECTIVE_ABSORB_KEY = "effectiveAbsorb";
51054
+ function absorbEnabled(config) {
51055
+ return config.absorb?.enabled === true;
51056
+ }
51057
+ function absorbToolName(config) {
51058
+ return config.absorb?.toolName ?? ABSORB_TOOL_NAME;
51059
+ }
51013
51060
  function effectiveAbsorbConfig(session, fallback) {
51014
51061
  const meta = session?.metadata[EFFECTIVE_ABSORB_KEY];
51015
51062
  if (meta && typeof meta === "object" && typeof meta.enabled === "boolean") {
@@ -51017,6 +51064,9 @@ function effectiveAbsorbConfig(session, fallback) {
51017
51064
  }
51018
51065
  return fallback.absorb;
51019
51066
  }
51067
+ function storeEffectiveAbsorb(session, config) {
51068
+ session.metadata[EFFECTIVE_ABSORB_KEY] = config.absorb ?? null;
51069
+ }
51020
51070
  function isProxyToolFor(name, session, config) {
51021
51071
  if (ACP_TOOL_NAMES.has(name)) return true;
51022
51072
  const absorb = effectiveAbsorbConfig(session, config);
@@ -51065,6 +51115,50 @@ function executeAbsorb(args, callId, absorb, ctx) {
51065
51115
  return outcome.resultText;
51066
51116
  }
51067
51117
 
51118
+ // src/acp-status.ts
51119
+ function handleAcpStatus(args, ctx) {
51120
+ const scope = typeof args.scope === "string" ? args.scope : void 0;
51121
+ const view = typeof args.view === "string" ? args.view : void 0;
51122
+ const tool = typeof args.tool === "string" ? args.tool : void 0;
51123
+ const sort = typeof args.sort === "string" ? args.sort : void 0;
51124
+ const limit = typeof args.limit === "number" ? args.limit : void 0;
51125
+ const base = buildStatusReport(ctx.session.state, ctx.messages, defaultCountTokens, { scope, view, tool, sort, limit });
51126
+ if (scope) return base;
51127
+ const extra = [];
51128
+ try {
51129
+ const turn = ctx.core.processTurn({
51130
+ messages: ctx.messages,
51131
+ state: ctx.session.state,
51132
+ config: ctx.config,
51133
+ tokenCount: ctx.session.stats.lastInputTokens,
51134
+ renderTags: "none"
51135
+ });
51136
+ const nudge = turn.nudge;
51137
+ if (nudge) {
51138
+ extra.push("");
51139
+ extra.push(nudge.shouldInject ? `Nudge: ACTIVE \u2014 ${nudge.reason}` : `Nudge: idle \u2014 ${nudge.reason}`);
51140
+ const ranges = viableRanges(nudge.compressibleRanges);
51141
+ const protectedRanges = nudge.protectedRanges ?? [];
51142
+ if (ranges.length > 0 || protectedRanges.length > 0) {
51143
+ extra.push("");
51144
+ extra.push(formatRanges(ranges, protectedRanges));
51145
+ }
51146
+ }
51147
+ } catch {
51148
+ }
51149
+ const archive = preCompactionArchiveOf(ctx.session);
51150
+ const archivedIds = Object.keys(archive);
51151
+ if (archivedIds.length > 0) {
51152
+ extra.push("");
51153
+ extra.push(`PRE-COMPACTION ARCHIVE \u2014 ${archivedIds.length} block(s): content was replaced by the client's native compaction summary, so it is no longer in the session history and decompress is unavailable.`);
51154
+ for (const id of archivedIds) {
51155
+ extra.push(` ${id} \u2014 ${archive[id].reason}`);
51156
+ }
51157
+ }
51158
+ return extra.length > 0 ? `${base}
51159
+ ${extra.join("\n")}` : base;
51160
+ }
51161
+
51068
51162
  // src/decompress-shared.ts
51069
51163
  import { mkdirSync as mkdirSync4, unlinkSync as unlinkSync2, writeFileSync as writeFileSync3 } from "fs";
51070
51164
  import { dirname as dirname2, join as join2 } from "path";
@@ -51885,24 +51979,24 @@ function costForUrl(url) {
51885
51979
  }
51886
51980
  return applyCap(REMOTE_IMAGE_TOKENS);
51887
51981
  }
51888
- function isObj(v2) {
51982
+ function isObj2(v2) {
51889
51983
  return typeof v2 === "object" && v2 !== null;
51890
51984
  }
51891
51985
  function urlOf(v2) {
51892
51986
  if (typeof v2 === "string") return v2;
51893
- if (isObj(v2) && typeof v2.url === "string") return v2.url;
51987
+ if (isObj2(v2) && typeof v2.url === "string") return v2.url;
51894
51988
  return void 0;
51895
51989
  }
51896
51990
  function imageTokensInParsedBody(protocol, body) {
51897
- if (!isObj(body)) return 0;
51991
+ if (!isObj2(body)) return 0;
51898
51992
  let total = 0;
51899
51993
  if (protocol === "responses") {
51900
51994
  const input = body.input;
51901
51995
  if (!Array.isArray(input)) return 0;
51902
51996
  for (const item of input) {
51903
- if (!isObj(item) || !Array.isArray(item.content)) continue;
51997
+ if (!isObj2(item) || !Array.isArray(item.content)) continue;
51904
51998
  for (const part of item.content) {
51905
- if (!isObj(part) || part.type !== "input_image") continue;
51999
+ if (!isObj2(part) || part.type !== "input_image") continue;
51906
52000
  const url = urlOf(part.image_url);
51907
52001
  if (url) total += costForUrl(url);
51908
52002
  }
@@ -51912,9 +52006,9 @@ function imageTokensInParsedBody(protocol, body) {
51912
52006
  const messages = body.messages;
51913
52007
  if (!Array.isArray(messages)) return 0;
51914
52008
  for (const m2 of messages) {
51915
- if (!isObj(m2) || !Array.isArray(m2.content)) continue;
52009
+ if (!isObj2(m2) || !Array.isArray(m2.content)) continue;
51916
52010
  for (const part of m2.content) {
51917
- if (!isObj(part)) continue;
52011
+ if (!isObj2(part)) continue;
51918
52012
  if (protocol === "openai") {
51919
52013
  if (part.type !== "image_url") continue;
51920
52014
  const url = urlOf(part.image_url);
@@ -51922,8 +52016,8 @@ function imageTokensInParsedBody(protocol, body) {
51922
52016
  } else {
51923
52017
  if (part.type !== "image") continue;
51924
52018
  const src = part.source;
51925
- if (isObj(src) && src.type === "base64" && typeof src.data === "string") total += applyCap(Math.ceil(src.data.length / 4));
51926
- else if (isObj(src) && src.type === "url" && typeof src.url === "string") total += costForUrl(src.url);
52019
+ if (isObj2(src) && src.type === "base64" && typeof src.data === "string") total += applyCap(Math.ceil(src.data.length / 4));
52020
+ else if (isObj2(src) && src.type === "url" && typeof src.url === "string") total += costForUrl(src.url);
51927
52021
  }
51928
52022
  }
51929
52023
  }
@@ -52611,6 +52705,8 @@ var MIN_USAGE = 0.9;
52611
52705
  var WINDOW_MS = 15 * 60 * 1e3;
52612
52706
  var MIN_EVENTS = 3;
52613
52707
  var MAX_TRACKED_SESSIONS = 512;
52708
+ var RETRACT_MARGIN_PCT = 0.03;
52709
+ var RETRACT_MIN_DELTA = 256;
52614
52710
  var states = /* @__PURE__ */ new Map();
52615
52711
  function resolveConfirmedLimit(session, model) {
52616
52712
  const md = session.metadata ?? {};
@@ -52626,6 +52722,38 @@ function resolveLearnedLimit(session, model) {
52626
52722
  const perModel = model ? resolveConfirmedLimit(session, model) ?? resolveSpeculativeLimit(session, model) : void 0;
52627
52723
  return perModel ?? resolveConfirmedLimit(session) ?? resolveSpeculativeLimit(session);
52628
52724
  }
52725
+ function retractStaleLearnedLimits(session, model) {
52726
+ const x = session.stats?.lastInputTokens ?? 0;
52727
+ if (!(x > 0)) return false;
52728
+ const stale = (v2) => typeof v2 === "number" && v2 > 0 && x - v2 >= Math.max(RETRACT_MIN_DELTA, v2 * RETRACT_MARGIN_PCT);
52729
+ const md = session.metadata ?? {};
52730
+ const removed = [];
52731
+ const cm = md.confirmedContextLimits;
52732
+ const lm = md.learnedContextLimits;
52733
+ if (model) {
52734
+ if (cm && stale(cm[model])) {
52735
+ removed.push(`confirmed ${cm[model]}`);
52736
+ delete cm[model];
52737
+ }
52738
+ if (lm && stale(lm[model])) {
52739
+ removed.push(`learned ${lm[model]}`);
52740
+ delete lm[model];
52741
+ }
52742
+ }
52743
+ if (stale(md.confirmedContextLimit)) {
52744
+ removed.push(`confirmed ${String(md.confirmedContextLimit)}`);
52745
+ delete md.confirmedContextLimit;
52746
+ }
52747
+ if (stale(md.learnedContextLimit)) {
52748
+ removed.push(`learned ${String(md.learnedContextLimit)}`);
52749
+ delete md.learnedContextLimit;
52750
+ }
52751
+ if (removed.length === 0) return false;
52752
+ session.metadata = md;
52753
+ log("warn", `[${session.id}] retracted stale context window(s) [${removed.join(", ")}] for ${model ?? "(unknown model)"} \u2014 a later turn succeeded at ${x} input tokens above them; using the configured window again`);
52754
+ markDirty(session);
52755
+ return true;
52756
+ }
52629
52757
  function resolvedWindow2(session, model) {
52630
52758
  const md = session.metadata ?? {};
52631
52759
  const learned = resolveLearnedLimit(session, model);
@@ -55195,6 +55323,35 @@ data: ${JSON.stringify({ type: "message_delta", delta: { stop_reason: "end_turn"
55195
55323
  safeWrite(res, `event: message_stop
55196
55324
  data: ${JSON.stringify({ type: "message_stop" })}
55197
55325
 
55326
+ `);
55327
+ }
55328
+ } catch {
55329
+ } finally {
55330
+ try {
55331
+ res.end();
55332
+ } catch {
55333
+ }
55334
+ }
55335
+ }
55336
+ function emitPreflightError(res, protocol, error, log2) {
55337
+ const err2 = { type: "server_error", code: "preflight_compress_failed", message: error.message, retryable: error.retryable };
55338
+ log2?.(`[acp-proxy: preflight failed after early response commit \u2014 delivering in-band: ${error.message}]`);
55339
+ try {
55340
+ if (protocol === "openai") {
55341
+ safeWrite(res, `data: ${JSON.stringify({ error: err2 })}
55342
+
55343
+ data: [DONE]
55344
+
55345
+ `);
55346
+ } else if (protocol === "responses") {
55347
+ safeWrite(res, `event: error
55348
+ data: ${JSON.stringify({ type: "error", code: err2.code, message: err2.message })}
55349
+
55350
+ `);
55351
+ } else {
55352
+ safeWrite(res, `event: error
55353
+ data: ${JSON.stringify({ type: "error", error: { type: "server_error", code: err2.code, message: err2.message } })}
55354
+
55198
55355
  `);
55199
55356
  }
55200
55357
  } catch {
@@ -57259,7 +57416,8 @@ function tunnelAllowlistFromEnv(env = process.env) {
57259
57416
  }
57260
57417
 
57261
57418
  // src/content-encoding.ts
57262
- import { createBrotliDecompress, createGunzip, createInflate, createInflateRaw } from "zlib";
57419
+ import { promisify } from "util";
57420
+ import { brotliDecompress, gunzip, inflate, inflateRaw } from "zlib";
57263
57421
 
57264
57422
  // node_modules/fzstd/esm/index.mjs
57265
57423
  var ab = ArrayBuffer;
@@ -57932,54 +58090,31 @@ var Decompress = /* @__PURE__ */ (function() {
57932
58090
  })();
57933
58091
 
57934
58092
  // src/content-encoding.ts
57935
- var DecompressedTooLargeError = class extends Error {
57936
- constructor(limit) {
57937
- super(`decompressed request exceeds ${limit} bytes`);
57938
- this.limit = limit;
57939
- this.name = "DecompressedTooLargeError";
57940
- }
57941
- limit;
57942
- };
57943
- function streamDecode(factory, input, max) {
57944
- return new Promise((resolve, reject) => {
57945
- const chunks = [];
57946
- let size = 0;
57947
- const decoder = factory();
57948
- decoder.on("data", (chunk) => {
57949
- size += chunk.length;
57950
- if (size > max) {
57951
- decoder.destroy();
57952
- reject(new DecompressedTooLargeError(max));
57953
- return;
57954
- }
57955
- chunks.push(chunk);
57956
- });
57957
- decoder.on("end", () => resolve(Buffer.concat(chunks, size)));
57958
- decoder.on("error", (err2) => reject(err2));
57959
- decoder.end(input);
57960
- });
57961
- }
57962
- async function decodeZstd(input, max) {
58093
+ var gunzipAsync = promisify(gunzip);
58094
+ var inflateAsync = promisify(inflate);
58095
+ var inflateRawAsync = promisify(inflateRaw);
58096
+ var brotliAsync = promisify(brotliDecompress);
58097
+ async function decodeZstd(input, maxOutputBytes) {
57963
58098
  const chunks = [];
57964
58099
  let size = 0;
57965
58100
  const decoder = new Decompress((chunk) => {
57966
58101
  size += chunk.byteLength;
57967
- if (size > max) throw new DecompressedTooLargeError(max);
58102
+ if (size > maxOutputBytes) throw new Error(`decompressed request exceeds ${maxOutputBytes} bytes`);
57968
58103
  chunks.push(Buffer.from(chunk));
57969
58104
  });
57970
58105
  decoder.push(input, true);
57971
58106
  return Buffer.concat(chunks, size);
57972
58107
  }
57973
- async function decodeOne(coding, input, max) {
57974
- if (coding === "gzip" || coding === "x-gzip") return streamDecode(createGunzip, input, max);
57975
- if (coding === "br") return streamDecode(createBrotliDecompress, input, max);
57976
- if (coding === "zstd") return decodeZstd(input, max);
58108
+ async function decodeOne(coding, input, maxOutputBytes) {
58109
+ const options = { maxOutputLength: maxOutputBytes };
58110
+ if (coding === "gzip" || coding === "x-gzip") return gunzipAsync(input, options);
58111
+ if (coding === "br") return brotliAsync(input, options);
58112
+ if (coding === "zstd") return decodeZstd(input, maxOutputBytes);
57977
58113
  if (coding === "deflate") {
57978
58114
  try {
57979
- return await streamDecode(createInflate, input, max);
57980
- } catch (err2) {
57981
- if (err2 instanceof DecompressedTooLargeError) throw err2;
57982
- return streamDecode(createInflateRaw, input, max);
58115
+ return await inflateAsync(input, options);
58116
+ } catch {
58117
+ return inflateRawAsync(input, options);
57983
58118
  }
57984
58119
  }
57985
58120
  throw new Error(`unsupported request content-encoding: ${coding}`);
@@ -57990,7 +58125,9 @@ async function decodeRequestBody(contentEncoding, input, maxOutputBytes) {
57990
58125
  let body = input;
57991
58126
  for (const coding of codings.reverse()) {
57992
58127
  body = await decodeOne(coding, body, maxOutputBytes);
57993
- if (body.byteLength > maxOutputBytes) throw new DecompressedTooLargeError(maxOutputBytes);
58128
+ if (body.byteLength > maxOutputBytes) {
58129
+ throw new Error(`decompressed request exceeds ${maxOutputBytes} bytes`);
58130
+ }
57994
58131
  }
57995
58132
  return { body, decoded: true };
57996
58133
  }
@@ -58043,6 +58180,7 @@ var LAUNCHER_MODEL_WINDOWS = parseLauncherModelWindows(process.env.BILI_LAUNCHER
58043
58180
  function launcherContextWindow(model) {
58044
58181
  return LAUNCHER_MODEL_WINDOWS[model];
58045
58182
  }
58183
+ var windowSourceLogged = /* @__PURE__ */ new Set();
58046
58184
  function anthropicBetaContextWindow(headers) {
58047
58185
  const raw = headers["anthropic-beta"];
58048
58186
  if (raw === void 0) return void 0;
@@ -58569,24 +58707,18 @@ async function handle(req, res, opts, core, config, log2, instanceId, instanceSt
58569
58707
  upstreamOrigin = route ? route.upstream : /^https?:\/\//i.test(url) ? new URL(url).origin : opts.upstream;
58570
58708
  protocol = route?.explicitProtocol ?? (req.method === "POST" && bodyBuffer.length > 0 ? urlPath.endsWith("/chat/completions") ? "openai" : urlPath.endsWith("/v1/messages") || urlPath.endsWith("/messages") ? "anthropic" : urlPath.endsWith("/responses") || responsesCompact ? "responses" : null : null);
58571
58709
  if (protocol !== null && bodyBuffer.length > 0) {
58572
- try {
58573
- const decoded = await decodeRequestBody(headerValue2(req, "content-encoding"), bodyBuffer, MAX_REQUEST_BYTES);
58574
- bodyBuffer = decoded.body;
58575
- if (decoded.decoded) delete req.headers["content-encoding"];
58576
- } catch (decErr) {
58577
- if (decErr instanceof DecompressedTooLargeError) throw decErr;
58578
- protocol = null;
58579
- log2("warn", `decode body failed (${String(decErr)}) - forwarding raw body verbatim to ${upstreamOrigin}`);
58580
- }
58710
+ const decoded = await decodeRequestBody(headerValue2(req, "content-encoding"), bodyBuffer, MAX_REQUEST_BYTES);
58711
+ bodyBuffer = decoded.body;
58712
+ if (decoded.decoded) delete req.headers["content-encoding"];
58581
58713
  }
58582
58714
  } catch (err2) {
58583
- if (err2 instanceof BodyTooLargeError || err2 instanceof DecompressedTooLargeError) {
58715
+ if (err2 instanceof BodyTooLargeError) {
58584
58716
  log2("warn", `413: request body exceeds ${err2.limit} bytes`);
58585
58717
  res.writeHead(413, { "content-type": "application/json" });
58586
58718
  res.end(JSON.stringify({ error: { type: "request_too_large", message: err2.message } }));
58587
58719
  return;
58588
58720
  }
58589
- log2("warn", `failed to prepare inbound request (${String(err2)}) - 400`);
58721
+ log2("warn", `read/decode body failed: ${String(err2)}`);
58590
58722
  res.writeHead(400, { "content-type": "application/json" });
58591
58723
  res.end(JSON.stringify({ error: { type: "invalid_request", message: String(err2) } }));
58592
58724
  return;
@@ -58657,6 +58789,13 @@ ${bodyBuffer.toString("utf8")}`);
58657
58789
  if (native) nativeFromFallback = false;
58658
58790
  }
58659
58791
  reqConfig = resolveRequestConfig(config, opts.routes, embeddedUrl, model, native, opts.compress);
58792
+ {
58793
+ if (!windowSourceLogged.has(model)) {
58794
+ windowSourceLogged.add(model);
58795
+ const wsSource = betaWindow ? "anthropic-beta" : pluginWindow ? "plugin" : launcherWindow ? "launcher" : configuredWindow ? "configured" : peekWindow ? "registry-peek" : native ? "table-or-registry" : "default";
58796
+ log2("info", `[window] model=${model} source=${wsSource} native=${native ?? "none"} effective=${reqConfig.modelContextLimit} launcher=${launcherWindow ?? "none"} configured=${configuredWindow ?? "none"} peek=${peekWindow ?? "none"} fallback=${nativeFromFallback}`);
58797
+ }
58798
+ }
58660
58799
  const aligned = operatorWindowTuned ? { limit: reqConfig.modelContextLimit, clamped: false } : codexAlignedWindow(reqConfig.modelContextLimit, model, req.headers);
58661
58800
  if (aligned.clamped) {
58662
58801
  const before = reqConfig.modelContextLimit;
@@ -58765,8 +58904,7 @@ ${bodyBuffer.toString("utf8")}`);
58765
58904
  restoreOutputBudget(parsed, session, log2);
58766
58905
  if (!countTokens && !responsesCompact && protocol !== null && isSideRequest(parsed)) {
58767
58906
  const reqModel2 = parsed.model;
58768
- const learnedMap2 = session.metadata.learnedContextLimits;
58769
- const learnedLimit2 = (reqModel2 && learnedMap2 ? learnedMap2[reqModel2] : void 0) ?? session.metadata.learnedContextLimit;
58907
+ const learnedLimit2 = resolveLearnedLimit(session, reqModel2);
58770
58908
  const guard = sideRequestGuard(parsed, protocol, reqConfig.modelContextLimit, learnedLimit2);
58771
58909
  if (guard.blocked) {
58772
58910
  log2("warn", `[${session.id}] side request (~${guard.estimate} tokens) \u2265 effective window ${guard.limit} (model=${reqModel2 ?? "?"}) \u2014 NOT forwarded: guaranteed upstream 400 (side requests bypass preflight by design, #388)`);
@@ -58798,23 +58936,24 @@ ${bodyBuffer.toString("utf8")}`);
58798
58936
  return;
58799
58937
  }
58800
58938
  const reqModel = parsed.model;
58801
- const learnedMap = session.metadata.learnedContextLimits;
58802
- const learnedLimit = (reqModel && learnedMap ? learnedMap[reqModel] : void 0) ?? session.metadata.learnedContextLimit;
58939
+ retractStaleLearnedLimits(session, reqModel);
58940
+ const confirmedLimit = resolveConfirmedLimit(session, reqModel);
58941
+ const learnedLimit = confirmedLimit ?? resolveSpeculativeLimit(session, reqModel);
58803
58942
  if (learnedLimit && learnedLimit > 0 && learnedLimit < reqConfig.modelContextLimit) {
58804
58943
  const resolved = reqConfig.modelContextLimit;
58805
58944
  reqConfig = { ...reqConfig, modelContextLimit: learnedLimit };
58806
58945
  nativeFromFallback = false;
58807
- log2("info", `[${session.id}] self-healed context window: ${resolved} \u2192 ${learnedLimit} (learned from an upstream overflow)`);
58946
+ log2("info", `[${session.id}] self-healed context window: ${resolved} \u2192 ${learnedLimit} (${confirmedLimit !== void 0 ? "confirmed by an upstream overflow error" : "weak-overflow heuristic"})`);
58808
58947
  } else if (nativeFromFallback && reqModel) {
58809
58948
  const prevInput = session.stats.lastInputTokens ?? 0;
58810
58949
  const prevWindow = session.metadata.lastTurnWindow;
58811
58950
  const resolved = reqConfig.modelContextLimit;
58812
58951
  if (prevWindow !== void 0 && prevInput > prevWindow && prevInput > resolved && prevInput >= 1e3) {
58813
- const map = session.metadata.learnedContextLimits ?? {};
58952
+ const map = session.metadata.confirmedContextLimits ?? {};
58814
58953
  const prev = map[reqModel];
58815
58954
  if (prev === void 0 || prevInput > prev) {
58816
58955
  map[reqModel] = prevInput;
58817
- session.metadata.learnedContextLimits = map;
58956
+ session.metadata.confirmedContextLimits = map;
58818
58957
  markDirty(session);
58819
58958
  }
58820
58959
  reqConfig = { ...reqConfig, modelContextLimit: prevInput };
@@ -58839,7 +58978,16 @@ ${bodyBuffer.toString("utf8")}`);
58839
58978
  acquireInFlight(session);
58840
58979
  try {
58841
58980
  await withSessionLock(session, async () => {
58842
- const runPrepare = () => countTokens ? prepareCountTokens(parsed, core, reqConfig, log2, session) : protocol === "anthropic" ? prepareAnthropic(parsed, req, opts, core, reqConfig, reqPrompts, log2, session, pluginMode) : protocol === "openai" ? prepareOpenai(parsed, req, opts, core, reqConfig, reqPrompts, log2, session, pluginMode, nativeWindow) : responsesCompact ? prepareResponsesCompact(bodyBuffer, parsed, session, req, core, reqConfig, log2) : prepareResponses(parsed, req, opts, core, reqConfig, reqPrompts, log2, session, responsesIdentity, pluginMode, upstreamOrigin, nativeWindow);
58981
+ const runPrepare = () => {
58982
+ const cs2 = resolveCompress(opts.routes, route?.rewrittenUrl, parsed.model, opts.compress);
58983
+ const keepRecent = cs2.stripImagesKeepRecent ?? DEFAULT_STRIP_IMAGES_KEEP_RECENT;
58984
+ const stripped = cs2.stripImages ? stripHistoricalImages(parsed, protocol, keepRecent) : { body: parsed, removed: 0 };
58985
+ if (opts.debug && stripped.removed > 0) {
58986
+ log2("info", `[debug] strip-images: dropped ${stripped.removed} historical image part(s), kept last ${keepRecent} (session=${session.id})`);
58987
+ }
58988
+ const work = stripped.body;
58989
+ return countTokens ? prepareCountTokens(work, core, reqConfig, log2, session) : protocol === "anthropic" ? prepareAnthropic(work, req, opts, core, reqConfig, reqPrompts, log2, session, pluginMode) : protocol === "openai" ? prepareOpenai(work, req, opts, core, reqConfig, reqPrompts, log2, session, pluginMode, nativeWindow) : responsesCompact ? prepareResponsesCompact(stripped.removed > 0 ? Buffer.from(JSON.stringify(work)) : bodyBuffer, work, session, req, core, reqConfig, log2) : prepareResponses(work, req, opts, core, reqConfig, reqPrompts, log2, session, responsesIdentity, pluginMode, upstreamOrigin, nativeWindow);
58990
+ };
58843
58991
  const isCodexCompactTrigger = protocol === "responses" && !responsesCompact && isCodexClient(req.headers) && hasCompactionTrigger(parsed.input);
58844
58992
  if (isCodexCompactTrigger) {
58845
58993
  const mode = codexCompactMode();
@@ -58868,23 +59016,42 @@ ${bodyBuffer.toString("utf8")}`);
58868
59016
  parsed.model,
58869
59017
  route,
58870
59018
  affinity,
59019
+ anonAffinity !== null,
58871
59020
  log2,
58872
59021
  instanceId
58873
59022
  );
58874
59023
  if (isPreflightFailFast(outcome)) {
58875
- if (outcome.respond && !res.headersSent && !res.destroyed) {
58876
- res.writeHead(outcome.status, {
58877
- "content-type": "application/json",
58878
- ...outcome.status === 503 ? { "retry-after": "30" } : {}
58879
- });
58880
- res.end(JSON.stringify({
58881
- error: {
58882
- type: "server_error",
58883
- code: "preflight_compress_failed",
58884
- message: outcome.message,
58885
- retryable: outcome.retryable
59024
+ if (outcome.respond && !res.destroyed) {
59025
+ if (res.headersSent) {
59026
+ if (prepared.stream) {
59027
+ emitPreflightError(res, prepared.protocol, { message: outcome.message, retryable: outcome.retryable }, (m2) => log2("warn", m2));
59028
+ } else {
59029
+ try {
59030
+ res.end(JSON.stringify({
59031
+ error: {
59032
+ type: "server_error",
59033
+ code: "preflight_compress_failed",
59034
+ message: outcome.message,
59035
+ retryable: outcome.retryable
59036
+ }
59037
+ }));
59038
+ } catch {
59039
+ }
58886
59040
  }
58887
- }));
59041
+ } else {
59042
+ res.writeHead(outcome.status, {
59043
+ "content-type": "application/json",
59044
+ ...outcome.status === 503 ? { "retry-after": "30" } : {}
59045
+ });
59046
+ res.end(JSON.stringify({
59047
+ error: {
59048
+ type: "server_error",
59049
+ code: "preflight_compress_failed",
59050
+ message: outcome.message,
59051
+ retryable: outcome.retryable
59052
+ }
59053
+ }));
59054
+ }
58888
59055
  }
58889
59056
  return;
58890
59057
  }
@@ -58997,12 +59164,17 @@ function diagNudge(turn, sessionId, tokenCount, limit, model, willInject) {
58997
59164
  }
58998
59165
  function armHostUsageCredit(session, originalMessages, processedMessages, log2) {
58999
59166
  session.hostCreditTokens = 0;
59000
- if (session.metadata.pluginAgent === "pi") return;
59167
+ if (session.metadata.pluginAgent === "pi" || session.metadata.pluginAgent === "omp") return;
59001
59168
  session.hostCreditTokens = processedMessages.length > 0 ? Math.max(0, estimateCoreMessages(originalMessages) - estimateCoreMessages(processedMessages)) : 0;
59002
59169
  if (session.hostCreditTokens > 0) {
59003
59170
  log2("info", `[${session.id}] host usage backfill armed: +${session.hostCreditTokens} tok (forwarded view is folded); host usage will report the uncompressed baseline`);
59004
59171
  }
59005
59172
  }
59173
+ function effectiveTokenCount(session, msgs) {
59174
+ if (session.stats.lastInputTokens > 0) return session.stats.lastInputTokens;
59175
+ if (!session.metadata.anonymousPrefixAffinity) return 0;
59176
+ return estimateCoreMessagesUpper(msgs);
59177
+ }
59006
59178
  function prepareAnthropic(parsed, req, opts, core, config, prompts, log2, session, pluginMode) {
59007
59179
  const sessionId = session.id;
59008
59180
  const stream2 = parsed.stream === true;
@@ -59027,11 +59199,15 @@ function prepareAnthropic(parsed, req, opts, core, config, prompts, log2, sessio
59027
59199
  const { msgs, cacheControls } = anthropicToCore(parsed);
59028
59200
  originalMessages = msgs;
59029
59201
  extractSystem(parsed.system);
59030
- const tokenCount = session.stats.lastInputTokens;
59202
+ const tokenCount = effectiveTokenCount(session, msgs);
59031
59203
  const activeBefore = new Set(session.state.blocks.filter((b2) => b2.active).map((b2) => b2.blockId));
59032
- const turn = core.processTurn({ messages: msgs, state: session.state, config, tokenCount, renderTags: "text-only" });
59204
+ const absorbActive = absorbEnabled(config) && opts.compress.injectTool;
59205
+ const loopConfig = absorbActive ? config : { ...config, absorb: void 0 };
59206
+ const turn = core.processTurn({ messages: msgs, state: session.state, config: loopConfig, tokenCount, renderTags: "text-only" });
59033
59207
  session.state = turn.state;
59034
59208
  session.stats.compressCreditTokens = 0;
59209
+ storeEffectiveAbsorb(session, loopConfig);
59210
+ turn.messages = applyAbsorbView(turn.messages, session.state, loopConfig, tokenCount);
59035
59211
  if (turn.nudge) turn.nudge.compressibleRanges = viableRanges(turn.nudge.compressibleRanges);
59036
59212
  nudge = turn.nudge;
59037
59213
  session.stats.contextTokens = tokenCount;
@@ -59046,9 +59222,9 @@ function prepareAnthropic(parsed, req, opts, core, config, prompts, log2, sessio
59046
59222
  applyCompactionArchive(session, activeBefore, new Set(msgs.map((m2) => m2.id)), log2);
59047
59223
  reapOrphanBlocks(session, msgs, deactivateBlock);
59048
59224
  rebuiltMessages = coreToAnthropic(processedMessages, cacheControls);
59049
- systemOut = injectSystem(parsed, opts, prompts);
59225
+ systemOut = injectSystem(parsed, opts, prompts, loopConfig);
59050
59226
  if (injectTools) {
59051
- toolsOut = injectTool(parsed.tools);
59227
+ toolsOut = injectTool(parsed.tools, absorbActive ? ABSORB_TOOL : void 0);
59052
59228
  }
59053
59229
  if (willInjectNudge && turn.nudge) {
59054
59230
  try {
@@ -59148,11 +59324,15 @@ function prepareOpenai(parsed, req, opts, core, config, prompts, log2, session,
59148
59324
  const { msgs, systemText } = openaiToCore(parsed);
59149
59325
  openaiSystemText = systemText;
59150
59326
  originalMessages = msgs;
59151
- const tokenCount = session.stats.lastInputTokens;
59327
+ const tokenCount = effectiveTokenCount(session, msgs);
59152
59328
  const activeBefore = new Set(session.state.blocks.filter((b2) => b2.active).map((b2) => b2.blockId));
59153
- const turn = core.processTurn({ messages: msgs, state: session.state, config, tokenCount, renderTags: "text-only" });
59329
+ const absorbActive = absorbEnabled(config) && shouldInject;
59330
+ const loopConfig = absorbActive ? config : { ...config, absorb: void 0 };
59331
+ const turn = core.processTurn({ messages: msgs, state: session.state, config: loopConfig, tokenCount, renderTags: "text-only" });
59154
59332
  session.state = turn.state;
59155
59333
  session.stats.compressCreditTokens = 0;
59334
+ storeEffectiveAbsorb(session, loopConfig);
59335
+ turn.messages = applyAbsorbView(turn.messages, session.state, loopConfig, tokenCount);
59156
59336
  if (turn.nudge) turn.nudge.compressibleRanges = viableRanges(turn.nudge.compressibleRanges);
59157
59337
  nudge = turn.nudge;
59158
59338
  session.stats.contextTokens = tokenCount;
@@ -59170,10 +59350,11 @@ function prepareOpenai(parsed, req, opts, core, config, prompts, log2, session,
59170
59350
  const sysParts = [];
59171
59351
  if (systemText) sysParts.push(systemText);
59172
59352
  if (shouldInject) sysParts.push(buildCompressSystemPrompt(prompts));
59353
+ if (absorbActive) sysParts.push(buildAbsorbSystemPrompt(absorbToolName(config)));
59173
59354
  rebuiltMessages = injectOpenaiSystem(rebuiltMessages, sysParts);
59174
59355
  openaiOutboundSystem = sysParts.join("\n\n");
59175
59356
  if (injectTools) {
59176
- toolsOut = injectOpenaiTool(parsed.tools);
59357
+ toolsOut = injectOpenaiTool(parsed.tools, absorbActive ? ABSORB_TOOL_OPENAI : void 0);
59177
59358
  }
59178
59359
  if (willInjectNudge && turn.nudge) {
59179
59360
  try {
@@ -59253,10 +59434,14 @@ function prepareResponses(parsed, req, opts, core, config, prompts, log2, sessio
59253
59434
  if (process.env.ACP_DEBUG) {
59254
59435
  log2("info", `[${sessionId}] input items: ${Array.isArray(parsed.input) ? parsed.input.map((i) => i.type).join(",") : "(string)"}`);
59255
59436
  }
59256
- const tokenCount = session.stats.lastInputTokens;
59257
- const turn = core.processTurn({ messages: msgs, state: session.state, config, tokenCount, renderTags });
59437
+ const tokenCount = effectiveTokenCount(session, msgs);
59438
+ const absorbActive = absorbEnabled(config) && shouldInject && !isCompactionTrigger && !responsesTextProtocol;
59439
+ const loopConfig = absorbActive ? config : { ...config, absorb: void 0 };
59440
+ const turn = core.processTurn({ messages: msgs, state: session.state, config: loopConfig, tokenCount, renderTags });
59258
59441
  session.state = turn.state;
59259
59442
  session.stats.compressCreditTokens = 0;
59443
+ storeEffectiveAbsorb(session, loopConfig);
59444
+ turn.messages = applyAbsorbView(turn.messages, session.state, loopConfig, tokenCount);
59260
59445
  if (turn.nudge) turn.nudge.compressibleRanges = viableRanges(turn.nudge.compressibleRanges);
59261
59446
  nudge = turn.nudge;
59262
59447
  session.stats.contextTokens = tokenCount;
@@ -59273,11 +59458,13 @@ function prepareResponses(parsed, req, opts, core, config, prompts, log2, sessio
59273
59458
  const forgedSummaries = echoReplaced ? [] : session.metadata.codexForgedSummaries ?? [];
59274
59459
  if (shouldInject && !isCompactionTrigger && !process.env.ACP_NO_COMPRESS_PROMPT) {
59275
59460
  const prompt = responsesTextProtocol ? buildCompressHybridSystemPrompt(prompts) : buildCompressSystemPrompt(prompts);
59276
- const devContent = [...projection.systemParts, ...forgedSummaries, prompt].join("\n\n---\n\n");
59461
+ const devParts = [...projection.systemParts, ...forgedSummaries, prompt];
59462
+ if (absorbActive) devParts.push(buildAbsorbSystemPrompt(absorbToolName(config)));
59463
+ const devContent = devParts.join("\n\n---\n\n");
59277
59464
  responsesDevContent = devContent;
59278
59465
  rebuiltInput = injectResponsesDeveloperMessage(rebuiltInput, devContent);
59279
59466
  if (!process.env.ACP_NO_INJECT_TOOL && injectTools) {
59280
- toolsOut = responsesTextProtocol ? injectResponsesTool(parsed.tools, ACP_READONLY_TOOLS_RESPONSES) : injectResponsesTool(parsed.tools);
59467
+ toolsOut = responsesTextProtocol ? injectResponsesTool(parsed.tools, ACP_READONLY_TOOLS_RESPONSES) : injectResponsesTool(parsed.tools, absorbActive ? [...ACP_TOOLS_RESPONSES, ABSORB_TOOL_RESPONSES] : ACP_TOOLS_RESPONSES);
59281
59468
  }
59282
59469
  } else if (projection.systemParts.length > 0 || forgedSummaries.length > 0) {
59283
59470
  const devContent = [...projection.systemParts, ...forgedSummaries].join("\n\n---\n\n");
@@ -59422,14 +59609,16 @@ function prepareResponsesCompact(body, parsed, session, req, core, config, log2)
59422
59609
  let transformOk = false;
59423
59610
  try {
59424
59611
  const projection = responsesToCore(forgeBody);
59425
- const turn = core.processTurn({ messages: projection.msgs, state: session.state, config, tokenCount: session.stats.lastInputTokens, renderTags: process.env.ACP_RENDER_NONE ? "none" : "text-only" });
59612
+ const compactConfig = { ...config, absorb: void 0 };
59613
+ const turn = core.processTurn({ messages: projection.msgs, state: session.state, config: compactConfig, tokenCount: session.stats.lastInputTokens, renderTags: process.env.ACP_RENDER_NONE ? "none" : "text-only" });
59426
59614
  session.state = turn.state;
59427
59615
  transformOk = true;
59428
59616
  if (!codexCompactGate(session, config.modelContextLimit, transformOk)) {
59429
59617
  session.state = prevState;
59430
59618
  return base;
59431
59619
  }
59432
- const processed = repairResponsesAssistantOrdering(stripKernelSummaries(turn.messages, turn.state), projection.msgs);
59620
+ const viewed = applyAbsorbView(turn.messages, turn.state, compactConfig, session.stats.lastInputTokens);
59621
+ const processed = repairResponsesAssistantOrdering(stripKernelSummaries(viewed, turn.state), projection.msgs);
59433
59622
  const output = patchResponsesInput(projection, processed);
59434
59623
  if (typeof output === "string") {
59435
59624
  session.state = prevState;
@@ -59465,10 +59654,11 @@ function isAutoModeClassifier(parsed) {
59465
59654
  if (!Array.isArray(stops)) return false;
59466
59655
  return stops.some((s3) => typeof s3 === "string" && AUTO_MODE_CLASSIFIER_STOPS.has(s3));
59467
59656
  }
59468
- function injectSystem(parsed, opts, prompts = defaultPrompts) {
59657
+ function injectSystem(parsed, opts, prompts = defaultPrompts, config) {
59469
59658
  const baseText = extractSystem(parsed.system);
59470
59659
  const parts = [];
59471
59660
  if (opts.compress.injectTool) parts.push(buildCompressSystemPrompt(prompts));
59661
+ if (opts.compress.injectTool && absorbEnabled(config)) parts.push(buildAbsorbSystemPrompt(absorbToolName(config)));
59472
59662
  if (parts.length === 0) return parsed.system;
59473
59663
  const full = baseText ? `${baseText}
59474
59664
 
@@ -59477,19 +59667,23 @@ function injectSystem(parsed, opts, prompts = defaultPrompts) {
59477
59667
  ${parts.join("\n\n")}` : parts.join("\n\n");
59478
59668
  return buildSystem(full, parsed.system);
59479
59669
  }
59480
- function injectTool(tools) {
59481
- if (!Array.isArray(tools)) return [...ACP_TOOLS_ANTHROPIC];
59670
+ function injectTool(tools, extra) {
59671
+ if (!Array.isArray(tools)) return extra ? [...ACP_TOOLS_ANTHROPIC, extra] : [...ACP_TOOLS_ANTHROPIC];
59482
59672
  const names = new Set(tools.map((t) => t?.name));
59483
59673
  const missing = ACP_TOOLS_ANTHROPIC.filter((t) => !names.has(t.name));
59484
- return missing.length === 0 ? tools : [...tools, ...missing];
59674
+ const extraMissing = extra && !names.has(extra.name);
59675
+ if (missing.length === 0 && !extraMissing) return tools;
59676
+ return [...tools, ...missing, ...extraMissing ? [extra] : []];
59485
59677
  }
59486
- function injectOpenaiTool(tools) {
59487
- if (!Array.isArray(tools)) return [...ACP_TOOLS_OPENAI];
59678
+ function injectOpenaiTool(tools, extra) {
59679
+ if (!Array.isArray(tools)) return extra ? [...ACP_TOOLS_OPENAI, extra] : [...ACP_TOOLS_OPENAI];
59488
59680
  const present = new Set(
59489
59681
  tools.map((t) => t?.function?.name).filter((n) => typeof n === "string")
59490
59682
  );
59491
59683
  const additions = ACP_TOOLS_OPENAI.filter((t) => !present.has(t.function.name));
59492
- return [...tools, ...additions];
59684
+ const out = [...tools, ...additions];
59685
+ if (extra && !out.some((t) => t?.function?.name === extra.function?.name)) out.push(extra);
59686
+ return out;
59493
59687
  }
59494
59688
  var FORCE_TEXT_PROTOCOL = process.env.ACP_COMPRESS_PROTOCOL === "text";
59495
59689
  function injectResponsesTool(tools, toolsToAdd = ACP_TOOLS_RESPONSES) {
@@ -59551,17 +59745,57 @@ function buildForwardTarget(req, opts, route, affinity, hopMarker) {
59551
59745
  function isPreflightFailFast(outcome) {
59552
59746
  return "failFast" in outcome;
59553
59747
  }
59554
- async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, core, config, model, route, affinity, log2, instanceId) {
59748
+ var PREFLIGHT_HOLD_GRACE_DEFAULT_MS = 3e4;
59749
+ var PREFLIGHT_KEEPALIVE_MS = 15e3;
59750
+ function preflightHoldGraceMs() {
59751
+ const raw = process.env.BILI_PREFLIGHT_HOLD_MS;
59752
+ if (!raw) return PREFLIGHT_HOLD_GRACE_DEFAULT_MS;
59753
+ const v2 = Number(raw);
59754
+ return Number.isFinite(v2) && v2 >= 0 ? Math.floor(v2) : PREFLIGHT_HOLD_GRACE_DEFAULT_MS;
59755
+ }
59756
+ function beginPreflightHold(res, prepared, log2) {
59757
+ if (res.headersSent || res.destroyed || res.writableEnded) return void 0;
59758
+ const sid = prepared.session.id;
59759
+ const keepAlive = prepared.stream ? ": bili-preflight\n\n" : " ";
59760
+ try {
59761
+ if (prepared.stream) {
59762
+ res.writeHead(200, {
59763
+ "content-type": "text/event-stream",
59764
+ "cache-control": "no-cache",
59765
+ "x-accel-buffering": "no",
59766
+ "x-bili-preflight": "compressing"
59767
+ });
59768
+ } else {
59769
+ res.writeHead(200, { "content-type": "application/json", "x-bili-preflight": "compressing" });
59770
+ }
59771
+ } catch {
59772
+ return void 0;
59773
+ }
59774
+ log2("info", `[${sid}] preflight still running after ${preflightHoldGraceMs()}ms grace \u2014 committed early ${prepared.stream ? "SSE" : "JSON"} headers + keep-alive to hold the client (#568)`);
59775
+ try {
59776
+ res.write(keepAlive);
59777
+ } catch {
59778
+ }
59779
+ const iv = setInterval(() => {
59780
+ try {
59781
+ res.write(keepAlive);
59782
+ } catch {
59783
+ clearInterval(iv);
59784
+ }
59785
+ }, PREFLIGHT_KEEPALIVE_MS);
59786
+ return () => clearInterval(iv);
59787
+ }
59788
+ async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, core, config, model, route, affinity, anonymous, log2, instanceId) {
59555
59789
  const session = prepared.session;
59556
59790
  const limit = config.modelContextLimit;
59557
59791
  const imageTokens = imageTokensInRawBody(prepared.protocol, prepared.body);
59558
59792
  const textEstimate = estimateCoreMessages(prepared.processedMessages);
59559
59793
  const overheadEstimate = estimateWireOverhead(prepared.protocol, prepared.body);
59560
59794
  const payloadEstimate = textEstimate + overheadEstimate + imageTokens;
59561
- const tokenCount = Math.max(session.stats.lastInputTokens, payloadEstimate);
59795
+ const unknownBaseline = anonymous && session.stats.lastInputTokens <= 0;
59796
+ const tokenCount = unknownBaseline ? estimateCoreMessagesUpper(prepared.processedMessages) + overheadEstimate + imageTokens : Math.max(session.stats.lastInputTokens, payloadEstimate);
59562
59797
  if (limit <= 0 || !model || tokenCount < limit) return prepared;
59563
- const learnedMap = session.metadata.learnedContextLimits;
59564
- const learnedLimit = (model ? learnedMap?.[model] : void 0) ?? session.metadata.learnedContextLimit;
59798
+ const learnedLimit = resolveLearnedLimit(session, model);
59565
59799
  const noOverflowEvidence = session.stats.lastInputTokens < limit && learnedLimit === void 0;
59566
59800
  if (imageTokens > 0 && textEstimate < limit && noOverflowEvidence) {
59567
59801
  log2("warn", `[${session.id}] image-dominated payload (~${textEstimate} text + ~${imageTokens} image tokens) exceeds window ${limit} by estimate only, no upstream overflow evidence \u2014 forwarding once so the upstream arbitrates billing (#496)`);
@@ -59573,9 +59807,14 @@ async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, c
59573
59807
  log2("error", `[${session.id}] preflight fail-fast ${status2} (retryable=${retryable}): ${message}`);
59574
59808
  return { failFast: true, status: status2, message, retryable, respond: !res.writableEnded };
59575
59809
  };
59576
- if ((prepared.nudge?.compressibleRanges ?? []).length === 0 && payloadEstimate < limit) {
59577
- log2("warn", `[${session.id}] preflight trigger fired on a stale baseline (~${tokenCount}) but the payload fits (~${payloadEstimate}/${limit}); forwarding as-is`);
59578
- return prepared;
59810
+ if ((prepared.nudge?.compressibleRanges ?? []).length === 0) {
59811
+ if (!unknownBaseline && payloadEstimate < limit) {
59812
+ log2("warn", `[${session.id}] preflight trigger fired on a stale baseline (~${tokenCount}) but the payload fits (~${payloadEstimate}/${limit}); forwarding as-is`);
59813
+ return prepared;
59814
+ }
59815
+ if (unknownBaseline) {
59816
+ return failFast(502, "no part of the conversation is compressible (nothing left to fold)", false);
59817
+ }
59579
59818
  }
59580
59819
  log2("warn", `[${session.id}] context ${tokenCount} tokens exceeds model window ${limit} (model=${model}); preflight compressing before forward`);
59581
59820
  const { upstreamUrl, headers, proxyUrl } = buildForwardTarget(req, opts, route, affinity, instanceId);
@@ -59584,30 +59823,43 @@ async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, c
59584
59823
  if (!res.writableEnded) clientAbort.abort();
59585
59824
  });
59586
59825
  const started = Date.now();
59587
- const result = await preflightCompress(
59588
- {
59589
- core,
59590
- session,
59591
- config,
59592
- prompts: prepared.prompts ?? defaultPrompts,
59593
- protocol: prepared.protocol,
59594
- url: upstreamUrl,
59595
- headers,
59596
- model,
59597
- proxyUrl,
59598
- signal: clientAbort.signal,
59599
- log: log2,
59600
- imageFloor: imageTokens,
59601
- wireOverhead: overheadEstimate
59602
- },
59603
- prepared.originalMessages
59604
- );
59826
+ let stopHold;
59827
+ const holdTimer = setTimeout(() => {
59828
+ stopHold = beginPreflightHold(res, prepared, log2);
59829
+ }, preflightHoldGraceMs());
59830
+ holdTimer.unref();
59831
+ let result;
59832
+ try {
59833
+ result = await preflightCompress(
59834
+ {
59835
+ core,
59836
+ session,
59837
+ config,
59838
+ prompts: prepared.prompts ?? defaultPrompts,
59839
+ protocol: prepared.protocol,
59840
+ url: upstreamUrl,
59841
+ headers,
59842
+ model,
59843
+ proxyUrl,
59844
+ signal: clientAbort.signal,
59845
+ log: log2,
59846
+ imageFloor: imageTokens,
59847
+ wireOverhead: overheadEstimate,
59848
+ unknownBaseline
59849
+ },
59850
+ prepared.originalMessages
59851
+ );
59852
+ } finally {
59853
+ clearTimeout(holdTimer);
59854
+ stopHold?.();
59855
+ }
59605
59856
  if (result.compressedRanges > 0) {
59606
59857
  log2("info", `[${session.id}] preflight compressed ${result.compressedRanges} range(s), ~${result.savedTokens} tokens saved (${tokenCount} \u2192 ${session.stats.lastInputTokens}) in ${Date.now() - started}ms; rebuilding payload`);
59607
59858
  const rebuilt = runPrepare();
59608
59859
  session.stats.requests -= 1;
59609
- if (estimateCoreMessages(rebuilt.processedMessages) + overheadEstimate + imageTokens < limit) return rebuilt;
59610
- } else if (estimateCoreMessages(prepared.processedMessages) + overheadEstimate + imageTokens < limit) {
59860
+ const fits = unknownBaseline ? result.fitsWindow : estimateCoreMessages(rebuilt.processedMessages) + overheadEstimate + imageTokens < limit;
59861
+ if (fits) return rebuilt;
59862
+ } else if (unknownBaseline ? result.fitsWindow : estimateCoreMessages(prepared.processedMessages) + overheadEstimate + imageTokens < limit) {
59611
59863
  log2("warn", `[${session.id}] preflight made no progress but the payload fits; forwarding as-is`);
59612
59864
  return prepared;
59613
59865
  }
@@ -59619,10 +59871,26 @@ async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, c
59619
59871
  const status = f2?.kind === "upstream" && f2.status === 429 ? 503 : 502;
59620
59872
  return failFast(status, f2?.detail ?? "the payload still exceeds the window after preflight compression", status === 503);
59621
59873
  }
59874
+ function armFailureShrink(prepared, log2, reason) {
59875
+ const s3 = prepared.session;
59876
+ let est;
59877
+ try {
59878
+ const text = typeof prepared.body === "string" ? prepared.body : prepared.body.toString("utf8");
59879
+ est = estimateTokensFast(text);
59880
+ } catch {
59881
+ return;
59882
+ }
59883
+ if (!Number.isFinite(est) || est <= 0) return;
59884
+ if (est > s3.stats.lastInputTokens) {
59885
+ s3.stats.lastInputTokens = est;
59886
+ markDirty(s3);
59887
+ log2("warn", `[${s3.id}] ${reason} with no usage report \u2014 armed emergency shrink with local estimate ${est} tokens`);
59888
+ }
59889
+ }
59622
59890
  async function forward(req, res, opts, body, prepared, core, config, log2, route, instanceId, affinity) {
59623
59891
  if (prepared?.codexForge) {
59624
59892
  log2("info", `[${prepared.session.id}] codex compact served locally (${prepared.codexForge.kind}); upstream not contacted`);
59625
- res.writeHead(200, { "content-type": prepared.codexForge.contentType });
59893
+ if (!res.headersSent) res.writeHead(200, { "content-type": prepared.codexForge.contentType });
59626
59894
  res.end(prepared.codexForge.body);
59627
59895
  return;
59628
59896
  }
@@ -59744,6 +60012,7 @@ ${bodyText}`);
59744
60012
  recordUpstreamConnection(upstreamUrl, proxyUrl);
59745
60013
  } catch (error) {
59746
60014
  recordUpstreamConnection(upstreamUrl, proxyUrl, error);
60015
+ if (prepared && req.method !== "GET" && req.method !== "HEAD") armFailureShrink(prepared, log2, "network failure");
59747
60016
  throw new Error(`upstream request failed: ${formatUpstreamError(error, upstreamUrl, proxyUrl)}`, { cause: error });
59748
60017
  }
59749
60018
  if (compatProtocol && typeof wireBody === "string" && upstreamResult.response.status === 400 && upstreamResult.response.body) {
@@ -59764,28 +60033,46 @@ ${bodyText}`);
59764
60033
  };
59765
60034
  const rejection = detectRoleRejection(upstreamResult.response.status, roleErrText);
59766
60035
  if (rejection && rejection.role !== "system") {
59767
- const fixed = applyCompatRoles(wireBody, compatProtocol, { [rejection.role]: "system" });
59768
- if (fixed.rewritten > 0) {
60036
+ const cp2 = compatProtocol;
60037
+ const wb = wireBody;
60038
+ const remember = (target, rewritten) => {
60039
+ const s3 = prepared?.session;
60040
+ if (s3) {
60041
+ const prev = s3.metadata.learnedCompatRoles ?? {};
60042
+ s3.metadata.learnedCompatRoles = { ...prev, [rejection.role]: target };
60043
+ markDirty(s3);
60044
+ }
60045
+ compatRoles = { ...compatRoles, [rejection.role]: target };
60046
+ const providerKey = new URL(upstreamUrl).origin;
60047
+ log2("info", `[${prepared?.session.id ?? "passthrough"}] [compat] upstream rejected role "${rejection.role}" \u2014 auto-rewrote ${rewritten} message role(s) to "${target}", retry OK (remembered for this session only). To make permanent, add: {"providers":{"${providerKey}":{"compat":{"roles":{"${rejection.role}":"${target}"}}}}`);
60048
+ };
60049
+ const hop = async (target) => {
60050
+ const fixed = applyCompatRoles(wb, cp2, { [rejection.role]: target });
60051
+ if (fixed.rewritten === 0) return "other";
60052
+ let r;
59769
60053
  try {
59770
- const retry = await fetchWithTimeout(upstreamUrl, { ...init, body: fixed.body }, void 0, clientAbort.signal);
59771
- if (retry.response.ok) {
59772
- upstreamResult.clearTimer();
59773
- const s3 = prepared?.session;
59774
- if (s3) {
59775
- const prev = s3.metadata.learnedCompatRoles ?? {};
59776
- s3.metadata.learnedCompatRoles = { ...prev, [rejection.role]: "system" };
59777
- markDirty(s3);
59778
- }
59779
- compatRoles = { ...compatRoles, [rejection.role]: "system" };
59780
- const providerKey = new URL(upstreamUrl).origin;
59781
- log2("info", `[${prepared?.session.id ?? "passthrough"}] [compat] upstream rejected role "${rejection.role}" \u2014 auto-rewrote ${fixed.rewritten} message role(s) to "system", retry OK (remembered for this session only). To make permanent, add: {"providers":{"${providerKey}":{"compat":{"roles":{"${rejection.role}":"system"}}}}`);
59782
- upstreamResult = retry;
59783
- } else {
59784
- retry.clearTimer();
59785
- }
60054
+ r = await fetchWithTimeout(upstreamUrl, { ...init, body: fixed.body }, void 0, clientAbort.signal);
59786
60055
  } catch {
60056
+ return "other";
59787
60057
  }
59788
- }
60058
+ if (r.response.ok) {
60059
+ upstreamResult.clearTimer();
60060
+ remember(target, fixed.rewritten);
60061
+ upstreamResult = r;
60062
+ return "ok";
60063
+ }
60064
+ let errText2 = null;
60065
+ if (r.response.body) {
60066
+ try {
60067
+ errText2 = (await readStreamToBuffer(r.response.body)).toString("utf8");
60068
+ } catch {
60069
+ errText2 = null;
60070
+ }
60071
+ }
60072
+ r.clearTimer();
60073
+ return errText2 !== null && detectSystemPlacementError(r.response.status, errText2) ? "placement-400" : "other";
60074
+ };
60075
+ if (await hop("system") === "placement-400") await hop("user");
59789
60076
  }
59790
60077
  }
59791
60078
  }
@@ -59843,30 +60130,37 @@ ${hdrText}
59843
60130
  } catch {
59844
60131
  reqModel = void 0;
59845
60132
  }
59846
- const learnedMap = s3.metadata.learnedContextLimits ?? {};
60133
+ const confirmedMap = s3.metadata.confirmedContextLimits ?? {};
59847
60134
  if (info.window) {
59848
- const prev = (reqModel ? learnedMap[reqModel] : void 0) ?? s3.metadata.learnedContextLimit;
59849
- if (reqModel) learnedMap[reqModel] = info.window;
59850
- else s3.metadata.learnedContextLimit = info.window;
59851
- s3.metadata.learnedContextLimits = learnedMap;
60135
+ const prev = (reqModel ? confirmedMap[reqModel] : void 0) ?? s3.metadata.confirmedContextLimit;
60136
+ if (reqModel) confirmedMap[reqModel] = info.window;
60137
+ else s3.metadata.confirmedContextLimit = info.window;
60138
+ s3.metadata.confirmedContextLimits = confirmedMap;
59852
60139
  log2("warn", `[${s3.id}] upstream context overflow \u2014 learned real window ${info.window} for ${reqModel ?? "(unknown model)"} (was ${prev ?? "unset"}); arming emergency shrink`);
59853
60140
  } else {
59854
60141
  const payloadEstimate = (prepared.processedMessages.length > 0 ? estimateCoreMessages(prepared.processedMessages) : estimateRawBodyTokens(parsedBody)) + rejectedImageTokens;
59855
- const prev = (reqModel ? learnedMap[reqModel] : void 0) ?? s3.metadata.learnedContextLimit;
60142
+ const prev = (reqModel ? confirmedMap[reqModel] : void 0) ?? s3.metadata.confirmedContextLimit;
59856
60143
  if (payloadEstimate >= 1e3 && (prev === void 0 || payloadEstimate < prev)) {
59857
- if (reqModel) learnedMap[reqModel] = payloadEstimate;
59858
- else s3.metadata.learnedContextLimit = payloadEstimate;
59859
- s3.metadata.learnedContextLimits = learnedMap;
60144
+ if (reqModel) confirmedMap[reqModel] = payloadEstimate;
60145
+ else s3.metadata.confirmedContextLimit = payloadEstimate;
60146
+ s3.metadata.confirmedContextLimits = confirmedMap;
59860
60147
  log2("warn", `[${s3.id}] upstream context overflow (window not parseable) \u2014 learned conservative window ${payloadEstimate} for ${reqModel ?? "(unknown model)"} from rejected payload size (was ${prev ?? "unset"}); arming emergency shrink`);
59861
60148
  } else {
59862
60149
  log2("warn", `[${s3.id}] upstream context overflow (window not parseable): ${info.message}`);
59863
60150
  }
59864
60151
  }
59865
- const floor = info.window ?? (reqModel ? learnedMap[reqModel] : void 0) ?? s3.metadata.learnedContextLimit ?? s3.metadata.effectiveContextLimit ?? 0;
59866
- if (floor > 0) s3.stats.lastInputTokens = Math.max(s3.stats.lastInputTokens, floor);
60152
+ if (info.window) {
60153
+ s3.stats.lastInputTokens = info.window;
60154
+ } else {
60155
+ const floor = (reqModel ? confirmedMap[reqModel] : void 0) ?? s3.metadata.confirmedContextLimit ?? s3.metadata.effectiveContextLimit ?? 0;
60156
+ if (floor > 0) s3.stats.lastInputTokens = Math.max(s3.stats.lastInputTokens, floor);
60157
+ }
59867
60158
  markDirty(s3);
59868
60159
  }
59869
60160
  }
60161
+ if (prepared?.session && upstream.status >= 500) {
60162
+ armFailureShrink(prepared, log2, `upstream ${upstream.status}`);
60163
+ }
59870
60164
  const errSid = prepared?.session.id ?? "unknown";
59871
60165
  const reqId = upstream.headers.get("x-request-id") ?? upstream.headers.get("request-id");
59872
60166
  const reqIdText = reqId ? ` request-id=${reqId}` : "";
@@ -59875,6 +60169,18 @@ ${hdrText}
59875
60169
  if (bodyText.length > 600) snippet += " \u2026";
59876
60170
  if (!snippet) snippet = "(no body)";
59877
60171
  log("warn", `[${errSid}] \u2190 upstream ${upstream.status}${reqIdText}: ${snippet}`);
60172
+ if (res.headersSent) {
60173
+ if (prepared?.stream) {
60174
+ emitStreamError(res, prepared.protocol, `upstream HTTP ${upstream.status}: ${snippet}`, (m2) => log("info", m2));
60175
+ } else {
60176
+ try {
60177
+ res.end(errBody ?? void 0);
60178
+ } catch {
60179
+ }
60180
+ }
60181
+ clearUpstreamTimer();
60182
+ return;
60183
+ }
59878
60184
  const errHeaders = { ...respHeaders };
59879
60185
  delete errHeaders["content-length"];
59880
60186
  delete errHeaders["transfer-encoding"];
@@ -59883,7 +60189,7 @@ ${hdrText}
59883
60189
  clearUpstreamTimer();
59884
60190
  return;
59885
60191
  }
59886
- res.writeHead(upstream.status, respHeaders);
60192
+ if (!res.headersSent) res.writeHead(upstream.status, respHeaders);
59887
60193
  if (!upstream.body) {
59888
60194
  res.end();
59889
60195
  clearUpstreamTimer();
@@ -59990,19 +60296,27 @@ ${hdrText}
59990
60296
  const parsedReq = JSON.parse(typeof body === "string" ? body : body.toString("utf8"));
59991
60297
  const reqHeaders = buildForwardHeaders(headers);
59992
60298
  const textProtocol = prepared.protocol === "responses" && !!prepared.responsesTextProtocol;
59993
- const systemPrompt = textProtocol ? buildCompressHybridSystemPrompt(prepared.prompts ?? defaultPrompts) : buildCompressSystemPrompt(prepared.prompts ?? defaultPrompts);
59994
- const adapter = pickAdapter(prepared.protocol, parsedReq, textProtocol, prepared.responsesProjection, prepared.anthropicSystem, prepared.openaiSystemText, prepared.session.hostCreditTokens ?? 0);
60299
+ const absorbActive = absorbEnabled(config) && opts.compress.injectTool && !textProtocol;
60300
+ const loopConfig = absorbActive ? config : { ...config, absorb: void 0 };
60301
+ const absorbSection = absorbActive ? `
60302
+
60303
+ ---
60304
+
60305
+ ${buildAbsorbSystemPrompt(absorbToolName(loopConfig))}` : "";
60306
+ const systemPrompt = (textProtocol ? buildCompressHybridSystemPrompt(prepared.prompts ?? defaultPrompts) : buildCompressSystemPrompt(prepared.prompts ?? defaultPrompts)) + absorbSection;
60307
+ const adapter = pickAdapter(prepared.protocol, parsedReq, textProtocol, prepared.responsesProjection, prepared.anthropicSystem, prepared.openaiSystemText, prepared.session.hostCreditTokens ?? 0, absorbActive ? absorbToolName(loopConfig) : void 0);
59995
60308
  const refreshFolded = (current) => {
59996
60309
  const turn = core.processTurn({
59997
60310
  messages: prepared.originalMessages,
59998
60311
  state: prepared.session.state,
59999
- config,
60312
+ config: loopConfig,
60000
60313
  tokenCount: prepared.session.stats.lastInputTokens,
60001
60314
  renderTags: prepared.renderTags ?? "text-only"
60002
60315
  });
60003
60316
  prepared.session.state = turn.state;
60317
+ const viewed = applyAbsorbView(turn.messages, turn.state, loopConfig, prepared.session.stats.lastInputTokens);
60004
60318
  const records = current.filter((m2) => typeof m2.id === "string" && m2.id.startsWith("acp_loop_"));
60005
- return repairResponsesAssistantOrdering(stripKernelSummaries([...turn.messages, ...records], turn.state), prepared.originalMessages);
60319
+ return repairResponsesAssistantOrdering(stripKernelSummaries([...viewed, ...records], turn.state), prepared.originalMessages);
60006
60320
  };
60007
60321
  const loop = runCompressLoop(
60008
60322
  streamToRead,