pi-mega-compact 0.21.10 → 0.21.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -302,6 +302,16 @@ export function loadConfig(): MegaConfig {
302
302
  contextHealthOutputQuality: envBool("MEGACOMPACT_CONTEXT_HEALTH_OUTPUT_QUALITY", true),
303
303
  contextHealthCachePoison: envBool("MEGACOMPACT_CONTEXT_HEALTH_CACHE_POISON", true),
304
304
  contextHealthMitigate: envBool("MEGACOMPACT_CONTEXT_HEALTH_MITIGATE", false),
305
+ // v0.21.12: invisible-overhead calibration — add the provider's fixed
306
+ // request overhead H (system+tools+extension prepends, never in the
307
+ // transcript) back into the token estimate for the headroom gate / tail
308
+ // cap. H is an EMA of observed wire samples per model, else
309
+ // wireOverheadDefaultPct × window. Default ON; OFF = byte-identical
310
+ // v0.21.11 (every H term is 0). Closes attempt #9 of the 32k overflow loop.
311
+ wireOverhead: envBool("MEGACOMPACT_WIRE_OVERHEAD", true),
312
+ // v0.21.12: fallback H as a fraction of the window when no EMA sample
313
+ // exists yet. Clamped [0, 0.85]; default 0.15. Percent-based.
314
+ wireOverheadDefaultPct: clamp(envFlag("MEGACOMPACT_WIRE_OVERHEAD_DEFAULT_PCT", 0.15), 0, 0.85),
305
315
  // D.1: env-overridable recompact delta (minimum context growth % before
306
316
  // re-compacting instead of replaying the cached live trim). Default 50.
307
317
  recompactPctDelta: envFlag("MEGACOMPACT_RECOMPACT_PCT_DELTA", 50),
@@ -16,9 +16,56 @@
16
16
  * Pure functions, no runtime dependency — trivially unit-testable headlessly.
17
17
  */
18
18
  import type { AgentMessage } from "@earendil-works/pi-agent-core";
19
- import { estimateBlockTokens, estimateMessageTokens } from "../../../src/tokens.js";
19
+ import { estimateBlockTokens } from "../../../src/tokens.js";
20
20
  import { messageContentText } from "./messageText.js";
21
21
 
22
+ /**
23
+ * Full-surface AgentMessage token estimate for BUDGET arithmetic (tail cap).
24
+ *
25
+ * convertToLlm (pi dist/core/messages.js) ships assistant/toolResult messages
26
+ * VERBATIM — every content block goes over the wire: text, thinking, toolCall
27
+ * (name + full `arguments` JSON), toolResult output, role wrappers. The text
28
+ * extractor (messageContentText) is lossy-on-purpose for analytics, and using
29
+ * it here made a GLM-4.7-style assistant message with ~11.6k bytes of toolCall
30
+ * arguments register as ~77 tokens — a 30k-token tail passed an 11.9k budget,
31
+ * the model overflowed, and pi's one-shot compact-and-retry failed
32
+ * ("Context overflow recovery failed", 2026-08-20 incident).
33
+ *
34
+ * Counts every byte the provider actually receives. Still a heuristic (len/4
35
+ * + 1 per block, like estimateBlockTokens) — just no longer lossy. Never
36
+ * throws: unknown block shapes fall back to their JSON serialization length,
37
+ * and a non-array/string content is counted as its serialization.
38
+ */
39
+ export function estimateAgentMessageBudgetTokens(m: AgentMessage): number {
40
+ try {
41
+ const c = (m as { content?: unknown }).content;
42
+ let bytes = 0;
43
+ if (typeof c === "string") {
44
+ bytes += c.length;
45
+ } else if (Array.isArray(c)) {
46
+ for (const b of c) {
47
+ if (b == null || typeof b !== "object") continue;
48
+ const o = b as Record<string, unknown>;
49
+ if (typeof o.text === "string") bytes += o.text.length;
50
+ if (typeof o.thinking === "string") bytes += o.thinking.length;
51
+ if (typeof o.name === "string") bytes += o.name.length;
52
+ if (o.arguments != null) bytes += JSON.stringify(o.arguments).length;
53
+ if (typeof o.output === "string") bytes += o.output.length;
54
+ // Per-block envelope overhead (role/type markers), matching the
55
+ // len/4+1 block accounting in estimateBlockTokens.
56
+ bytes += 4;
57
+ }
58
+ } else if (c != null) {
59
+ bytes += JSON.stringify(c).length;
60
+ }
61
+ return estimateBlockTokens(" ".repeat(Math.max(0, bytes)));
62
+ } catch {
63
+ // non-fatal: fall back to the legacy text-only estimate rather than
64
+ // disable the cap on a pathological message.
65
+ return estimateBlockTokens(messageContentText(m));
66
+ }
67
+ }
68
+
22
69
  /**
23
70
  * The model's declared maxTokens is only trusted as the output budget when it
24
71
  * is plausible. models.json carries sentinel junk for some entries (1e9,
@@ -112,6 +159,15 @@ export function applyTailCap(opts: {
112
159
  * otherwise each message's tokens are estimated from its text content.
113
160
  */
114
161
  messageTokens?: readonly number[];
162
+ /**
163
+ * v0.21.12: the provider's invisible overhead H (system prompt + tool
164
+ * definitions + extension systemPrompt prepends — everything pi adds at
165
+ * request time that NEVER appears in the stored transcript). Subtracted from
166
+ * the budget so the cap accounts for the REAL wire prompt, not just the
167
+ * counted messages. Default 0 (no H) — identical behavior to v0.21.11 when
168
+ * the wireOverhead flag is OFF. Clamped to >= 0.
169
+ */
170
+ overheadTokens?: number;
115
171
  }): { recent: AgentMessage[]; dropped: number } {
116
172
  const { recentRaw, summaryTokens, ctxWindow, outputReservePct } = opts;
117
173
  if (ctxWindow <= 0 || recentRaw.length <= 1) {
@@ -133,9 +189,16 @@ export function applyTailCap(opts: {
133
189
  // <= 95% of the window) plus margin + summary can still exceed the window
134
190
  // on tiny summaries-free edges; the floor keeps the cap alive with a small
135
191
  // positive budget instead of disabling it (pre-v0.21.9 behavior).
192
+ // v0.21.12: also subtract the invisible overhead H so the cap bounds the
193
+ // ACTUAL wire prompt (messages + H), breaking the 400 loop when the estimate
194
+ // undercounts by the system-prompt/tool-definition overhead.
136
195
  const budget = Math.max(
137
196
  1,
138
- ctxWindow - reserveTokens - safetyMargin - Math.max(0, summaryTokens),
197
+ ctxWindow -
198
+ reserveTokens -
199
+ safetyMargin -
200
+ Math.max(0, summaryTokens) -
201
+ Math.max(0, opts.overheadTokens ?? 0),
139
202
  );
140
203
  let start = 0;
141
204
  let tailTokens = 0;
@@ -143,7 +206,7 @@ export function applyTailCap(opts: {
143
206
  tailTokens +=
144
207
  msgTokens != null
145
208
  ? Math.max(0, msgTokens[i])
146
- : estimateMessageTokens({ text: messageContentText(recentRaw[i]) });
209
+ : estimateAgentMessageBudgetTokens(recentRaw[i]);
147
210
  if (tailTokens > budget) {
148
211
  // Keep from i+1 onward; never drop below the FINAL message.
149
212
  start = Math.min(i + 1, recentRaw.length - 1);
@@ -178,13 +241,16 @@ export function recapReplayedTail(opts: {
178
241
  maxOutputTokens: number;
179
242
  outputReservePct: number;
180
243
  safetyMarginPct: number;
244
+ /** v0.21.12: invisible overhead H (see applyTailCap). Default 0. */
245
+ overheadTokens?: number;
181
246
  }): { recent: AgentMessage[]; dropped: number } {
182
247
  return applyTailCap({
183
248
  recentRaw: opts.recentRaw,
184
- summaryTokens: estimateBlockTokens(messageContentText(opts.summaryAgentMsg)),
249
+ summaryTokens: estimateAgentMessageBudgetTokens(opts.summaryAgentMsg),
185
250
  ctxWindow: opts.ctxWindow,
186
251
  maxOutputTokens: opts.maxOutputTokens,
187
252
  outputReservePct: opts.outputReservePct,
188
253
  safetyMarginPct: opts.safetyMarginPct,
254
+ overheadTokens: opts.overheadTokens ?? 0,
189
255
  });
190
256
  }
@@ -45,6 +45,10 @@ export function buildLiveTrimView(
45
45
  ran: { result: CompactResult };
46
46
  perModelThreshold: { safetyMarginPct: number; firePointPct: number };
47
47
  tailResult: TailResultFn;
48
+ /** v0.21.12: the provider's invisible overhead H (system+tools+prepends),
49
+ * handler-resolved. Subtracted from the tail-cap budget so the cap bounds
50
+ * the REAL wire prompt. 0 when the flag is OFF (byte-identical v0.21.11). */
51
+ overheadTokens?: number;
48
52
  },
49
53
  ): { messages: AgentMessage[] } | undefined {
50
54
  const {
@@ -57,6 +61,7 @@ export function buildLiveTrimView(
57
61
  ran,
58
62
  perModelThreshold,
59
63
  tailResult,
64
+ overheadTokens = 0,
60
65
  } = opts;
61
66
 
62
67
  // S16 LIVE trim: collapse the compacted region to a summary + recent anchor.
@@ -105,6 +110,37 @@ export function buildLiveTrimView(
105
110
  anchorUserMessages,
106
111
  criticalOver: (pct ?? 0) >= 90,
107
112
  });
113
+ // v0.21.12: CAP THE SKIP PATH. When computeLiveTrimCut returns null
114
+ // (anchor floor blocked cutting a fat recent tool pair, or the
115
+ // criticalOver hatch stayed closed because estimated pressure ≈80%),
116
+ // the pre-v0.21.12 code shipped the RAW untrimmed view — which, with
117
+ // the invisible overhead H uncounted, overflowed the window → 400
118
+ // again, forever (the entire v0.21.11 blind spot). This is the
119
+ // invariant "the trim path never ships a view the budget wouldn't
120
+ // allow": even when we cannot summarize, we still front-drop
121
+ // OLDEST messages until the RAW tail + overhead fits the budget, so
122
+ // the model is never fed a prompt that exceeds its window. Flag OFF
123
+ // ⇒ byte-identical to v0.21.11 (return raw, no cap).
124
+ if (config.wireOverhead && runtime.lastCtxWindow > 0) {
125
+ const { recent, dropped } = applyTailCap({
126
+ recentRaw: messages,
127
+ summaryTokens: 0,
128
+ ctxWindow: runtime.lastCtxWindow,
129
+ maxOutputTokens: runtime.currentModel?.maxTokens ?? 0,
130
+ outputReservePct: config.outputReservePct,
131
+ safetyMarginPct: perModelThreshold.safetyMarginPct,
132
+ overheadTokens,
133
+ });
134
+ if (dropped > 0) {
135
+ runtime.diagCtxSkipCapped++;
136
+ runtime.logger.info("skip_cap_applied", {
137
+ sessionId: runtime.rt.sessionId,
138
+ dropped,
139
+ ctxWindow: runtime.lastCtxWindow,
140
+ });
141
+ return tailResult(recent) ?? { messages: recent };
142
+ }
143
+ }
108
144
  return tailResult() ?? undefined; // unsafe / below anchor floor — no trim this call
109
145
  }
110
146
  const summaryMsg = liveTrimSummaryMessage({
@@ -152,6 +188,7 @@ export function buildLiveTrimView(
152
188
  maxOutputTokens: runtime.currentModel?.maxTokens ?? 0,
153
189
  outputReservePct: config.outputReservePct,
154
190
  safetyMarginPct: modelThreshold.safetyMarginPct,
191
+ overheadTokens,
155
192
  });
156
193
  if (dropped > 0) {
157
194
  runtime.logger.warn("live-trim-tail-cap", {
@@ -48,6 +48,7 @@ export function invokePipeline(
48
48
  pct: number | null | undefined;
49
49
  currentTokens: number;
50
50
  tailResult: TailResultFn;
51
+ overheadTokens?: number;
51
52
  },
52
53
  ): PipelineOutcome {
53
54
  // VC5C: emit the rollout decision per compact event (observability seam).
@@ -133,6 +134,7 @@ export function invokePipeline(
133
134
  maxOutputTokens: runtime.currentModel?.maxTokens ?? 0,
134
135
  outputReservePct: config.outputReservePct,
135
136
  safetyMarginPct: runtime.trimCache.safetyMarginPct,
137
+ overheadTokens: opts.overheadTokens ?? 0,
136
138
  });
137
139
  runtime.diagLiveTrimFires++;
138
140
  runtime.diagLiveTrimReplays++;
@@ -0,0 +1,132 @@
1
+ /**
2
+ * context-handler/wireTruth.ts — invisible-overhead calibration + wire-truth parse.
3
+ *
4
+ * Attempt #9 on the small-context-model overflow loop (2026-08-19 incident,
5
+ * 8 prior attempts). ROOT CAUSE: pi adds a FIXED OVERHEAD H at request time
6
+ * (system prompt + tool definitions + extension systemPrompt prepends) that
7
+ * NEVER appears in the stored transcript. Neither pi's estimateContextTokens nor
8
+ * our estimateSessionTokens/applyTailCap count H. So when the provider 400s with
9
+ * "request (39048 tokens) exceeds the available context size (32768 tokens)" we
10
+ * have been estimating ~18–26k and judging headroom against that — undercounting
11
+ * by ~50%, so the gate never trips correctly and a RAW uncapped view ships → 400
12
+ * again, forever (pi's one-shot overflow recovery only resets on user
13
+ * message_start, so auto-retries burn it permanently).
14
+ *
15
+ * Two mechanisms here:
16
+ * - parseWireTruth: regex the provider's 400 text into ground-truth
17
+ * request/available token counts. Pure + unit-tested.
18
+ * - Per-model overhead EMA: calibrate H from observed samples so the gate and
19
+ * tail cap can ADD it back. Persisted in the SQLite meta table (copy the
20
+ * thrashGuard.ts pattern) so it survives restarts and travels with the repo.
21
+ *
22
+ * Non-fatal EVERYWHERE: every store read/write is best-effort, swallowed on
23
+ * failure. Structured JSON logging only (logger + emit dual-sink, mirroring
24
+ * armThrashGuard's 3WF-5 pattern). No console.*, no network, no mocks.
25
+ */
26
+ import { getMetaNumber, setMetaNumber } from "../../../src/store/sqlite.js";
27
+
28
+ /** EMA smoothing factor for the overhead calibration (fixed, no count needed). */
29
+ export const OVERHEAD_EMA_ALPHA = 0.4;
30
+
31
+ /** Safety clamp: an overhead sample may never exceed this fraction of the window. */
32
+ export const OVERHEAD_CLAMP_FRACTION = 0.85;
33
+
34
+ /** Meta key prefix; the model id is appended (meta stores integers only). */
35
+ export const OVERHEAD_META_PREFIX = "wire.overhead_ema.";
36
+
37
+ /**
38
+ * Parse the provider's overflow error text into ground-truth token counts.
39
+ *
40
+ * Matches strings like:
41
+ * "request (39048 tokens) exceeds the available context size (32768 tokens)"
42
+ * "request (39,048 tokens) exceeds the available context size (32,768 tokens)"
43
+ *
44
+ * Pure + side-effect free — trivially unit-testable. Returns null when the text
45
+ * does not match the expected shape (e.g. an unrelated error message).
46
+ */
47
+ export function parseWireTruth(
48
+ text: string,
49
+ ): { requestTokens: number; availableTokens: number } | null {
50
+ if (typeof text !== "string" || text.length === 0) return null;
51
+ const m = text.match(
52
+ /request\s*\((\d[\d,]*)\s*tokens\)\s*exceeds the available context size\s*\((\d[\d,]*)\s*tokens\)/i,
53
+ );
54
+ if (!m) return null;
55
+ const requestTokens = Number(m[1]?.replace(/,/g, ""));
56
+ const availableTokens = Number(m[2]?.replace(/,/g, ""));
57
+ if (!Number.isFinite(requestTokens) || !Number.isFinite(availableTokens)) return null;
58
+ if (requestTokens <= 0 || availableTokens <= 0) return null;
59
+ return { requestTokens, availableTokens };
60
+ }
61
+
62
+ /** Build the meta key for a model's overhead EMA (stored ×100, integer). */
63
+ function overheadKey(modelId: string): string {
64
+ return OVERHEAD_META_PREFIX + modelId;
65
+ }
66
+
67
+ /**
68
+ * Read the calibrated overhead (tokens) for a model. Returns 0 when absent or
69
+ * unreadable — non-fatal everywhere. When the context window is known, clamps the
70
+ * value into [0, 0.85 × ctxWindow] as a safety against a runaway EMA.
71
+ */
72
+ export function readWireOverhead(modelId: string, stateDir: string, ctxWindow = 0): number {
73
+ if (!modelId) return 0;
74
+ try {
75
+ const stored = getMetaNumber(overheadKey(modelId), stateDir);
76
+ if (!Number.isFinite(stored) || stored <= 0) return 0;
77
+ const frac = stored / 100;
78
+ let overhead = frac * (ctxWindow > 0 ? ctxWindow : 1);
79
+ if (ctxWindow > 0) {
80
+ overhead = Math.min(overhead, ctxWindow * OVERHEAD_CLAMP_FRACTION);
81
+ }
82
+ return Math.max(0, overhead);
83
+ } catch {
84
+ return 0; // non-fatal: never fail the agent loop on a store read
85
+ }
86
+ }
87
+
88
+ /**
89
+ * Fold a new overhead sample into the per-model EMA and persist it. Returns the
90
+ * new EMA in tokens (0 when the sample is invalid). First sample initializes the
91
+ * EMA directly (no warm-up). Best-effort: never throws; a store failure is
92
+ * swallowed and the in-memory EMA is still returned.
93
+ *
94
+ * Clamped into [0, 0.85 × ctxWindow] when ctxWindow is known.
95
+ */
96
+ export function sampleWireOverhead(
97
+ modelId: string,
98
+ stateDir: string,
99
+ sample: number,
100
+ ctxWindow = 0,
101
+ ): number {
102
+ if (!modelId) return 0;
103
+ if (!Number.isFinite(sample) || sample <= 0) return 0;
104
+ let ema = sample;
105
+ try {
106
+ const stored = getMetaNumber(overheadKey(modelId), stateDir);
107
+ if (Number.isFinite(stored) && stored > 0) {
108
+ // prev is stored as a fraction ×100; convert back to token space
109
+ // against the window before blending so the EMA is dimensionally
110
+ // consistent (token space in, token space out).
111
+ const prevTokens = (stored / 100) * (ctxWindow > 0 ? ctxWindow : sample);
112
+ ema = OVERHEAD_EMA_ALPHA * sample + (1 - OVERHEAD_EMA_ALPHA) * prevTokens;
113
+ }
114
+ // Convert EMA to a fraction of the window (so the meta value is window-
115
+ // independent + percent-based). When the window is unknown, clamp the
116
+ // stored fraction to a sane [0, OVERHEAD_CLAMP_FRACTION] band so a stale
117
+ // 0-window sample cannot poison a later windowed read.
118
+ let frac = ctxWindow > 0 ? ema / ctxWindow : ema;
119
+ if (ctxWindow > 0) {
120
+ frac = Math.min(frac, OVERHEAD_CLAMP_FRACTION);
121
+ } else {
122
+ frac = Math.min(Math.max(frac, 0), OVERHEAD_CLAMP_FRACTION);
123
+ }
124
+ setMetaNumber(overheadKey(modelId), Math.round(frac * 100), stateDir);
125
+ } catch {
126
+ /* non-fatal: best-effort meta write */
127
+ }
128
+ // Return the token-space EMA (clamped) regardless of whether the persist landed.
129
+ let out = ema;
130
+ if (ctxWindow > 0) out = Math.min(out, ctxWindow * OVERHEAD_CLAMP_FRACTION);
131
+ return Math.max(0, out);
132
+ }
@@ -0,0 +1,168 @@
1
+ /**
2
+ * context-handler/wireTruthApply.ts — v0.21.12 wired-overhead runtime seam.
3
+ *
4
+ * Delegate-shell sibling extracted from context-handler.ts (extensions/ 400-line
5
+ * soft limit). Holds the two wire-overhead call-site blocks that don't fit the
6
+ * handler's budget:
7
+ * - sampleWireOverheadFromUsage: calibrate the per-model overhead EMA from a
8
+ * healthy usage-bearing context event (H_sample = usage.tokens − the REAL
9
+ * message-list estimate). MUST use the true estimate, not the pct-derived
10
+ * fallback — otherwise hSample ≈ 0 and the EMA trains itself to nothing.
11
+ * - applyWireTruthOverride: when the last assistant message is an error whose
12
+ * text matches the provider's 400 shape, treat the parsed requestTokens as
13
+ * ground-truth currentTokens for THIS event's gate, feed the EMA, and prefer
14
+ * the parsed availableTokens over runtime.lastCtxWindow.
15
+ *
16
+ * Both are no-op / byte-identical when config.wireOverhead is OFF. Non-fatal
17
+ * everywhere; never throw on the agent loop. Structured logging only.
18
+ */
19
+ import type { AgentMessage } from "@earendil-works/pi-agent-core";
20
+ import type { MegaRuntime } from "../../mega-runtime.js";
21
+ import type { MegaConfig } from "../../mega-config.js";
22
+ import { parseWireTruth, sampleWireOverhead, readWireOverhead } from "./wireTruth.js";
23
+
24
+ /**
25
+ * Calibrate the overhead EMA from a usage-bearing context event. Called once per
26
+ * event (from the handler) when usage is finite + wireOverhead is ON.
27
+ * `estimateTokens` MUST be the REAL message-list estimate (the engineView
28
+ * estimate), never the pct-derived fallback — pi's percent reconstructs
29
+ * usage.tokens exactly, so using it would force hSample ≈ 0 and erase the EMA
30
+ * (0.6^5 retained after five healthy turns).
31
+ */
32
+ export function sampleWireOverheadFromUsage(opts: {
33
+ runtime: MegaRuntime;
34
+ config: MegaConfig;
35
+ modelId: string;
36
+ usageTokens: number | null | undefined;
37
+ resolvedWindow: number;
38
+ estimateTokens: number;
39
+ }): void {
40
+ const { runtime, config, modelId, usageTokens, resolvedWindow, estimateTokens } = opts;
41
+ if (!config.wireOverhead) return; // byte-identical when OFF
42
+ if (modelId === "" || usageTokens == null || !Number.isFinite(usageTokens) || resolvedWindow <= 0)
43
+ return;
44
+ const hSample = Math.max(0, usageTokens - estimateTokens);
45
+ try {
46
+ sampleWireOverhead(modelId, runtime.currentStateDir, hSample, resolvedWindow);
47
+ } catch {
48
+ /* non-fatal */
49
+ }
50
+ }
51
+
52
+ /**
53
+ * Resolve the invisible overhead H (tokens) the handler feeds to the tail-cap
54
+ * budget at the fire/replay paths. Returns the calibrated EMA (or the
55
+ * wireOverheadDefaultPct × window fallback when no sample exists yet). 0 when
56
+ * wireOverhead is OFF or no model/window is known — byte-identical to v0.21.11.
57
+ * Computed once so every call site passes the SAME H the gate used.
58
+ */
59
+ export function resolveOverheadTokens(opts: {
60
+ config: MegaConfig;
61
+ modelId: string;
62
+ resolvedWindow: number;
63
+ stateDir: string;
64
+ }): number {
65
+ const { config, modelId, resolvedWindow, stateDir } = opts;
66
+ if (!config.wireOverhead || modelId === "" || resolvedWindow <= 0) return 0;
67
+ const e = readWireOverhead(modelId, stateDir, resolvedWindow);
68
+ return e > 0 ? e : config.wireOverheadDefaultPct * resolvedWindow;
69
+ }
70
+
71
+ /**
72
+ * v0.21.12: invisible-overhead correction of the ESTIMATE-path token count. When
73
+ * the token estimate came from the message list (not provider usage) AND
74
+ * wireOverhead is ON, add H so the gate/thrash/tail-cap see the REAL request
75
+ * size. H defaults to wireOverheadDefaultPct × window until a wire sample
76
+ * calibrates it. Flag OFF ⇒ H = 0 (byte-identical to v0.21.11). The provider
77
+ * usage path is ground truth for the message-list size and is never corrected.
78
+ */
79
+ export function correctEstimateWithOverhead(opts: {
80
+ config: MegaConfig;
81
+ tokenSource: "usage" | "estimate" | "pct";
82
+ rawTokens: number;
83
+ modelId: string;
84
+ resolvedWindow: number;
85
+ stateDir: string;
86
+ }): number {
87
+ const { config, tokenSource, rawTokens, modelId, resolvedWindow, stateDir } = opts;
88
+ if (!config.wireOverhead || tokenSource !== "estimate" || resolvedWindow <= 0) return rawTokens;
89
+ const h = modelId !== "" ? readWireOverhead(modelId, stateDir, resolvedWindow) : 0;
90
+ const H = h > 0 ? h : config.wireOverheadDefaultPct * resolvedWindow;
91
+ return rawTokens + H;
92
+ }
93
+
94
+ /**
95
+ * Apply the wire-truth gate override for THIS event. Returns the (possibly
96
+ * corrected) currentTokens. When the last assistant message is an error
97
+ * (stopReason "error" or an errorMessage field) and its text matches the
98
+ * provider's overflow error shape, the parsed requestTokens become
99
+ * ground-truth currentTokens — even when our estimate reads far below every
100
+ * threshold. Also feeds the EMA and prefers the parsed availableTokens over
101
+ * runtime.lastCtxWindow when they differ. No-op (returns currentTokens
102
+ * unchanged) when wireOverhead is OFF or no error/parse matched.
103
+ */
104
+ export function applyWireTruthOverride(opts: {
105
+ runtime: MegaRuntime;
106
+ config: MegaConfig;
107
+ messages: readonly AgentMessage[];
108
+ modelId: string;
109
+ resolvedWindow: number;
110
+ estimateTokens: number;
111
+ currentTokens: number;
112
+ }): number {
113
+ const { runtime, config, messages, modelId, resolvedWindow, estimateTokens, currentTokens } = opts;
114
+ if (!config.wireOverhead) return currentTokens; // byte-identical when OFF
115
+ try {
116
+ const last = messages[messages.length - 1] as
117
+ | { role?: string; stopReason?: string; errorMessage?: string; content?: unknown }
118
+ | undefined;
119
+ const lastText =
120
+ typeof last?.errorMessage === "string"
121
+ ? last.errorMessage
122
+ : typeof last?.content === "string"
123
+ ? last.content
124
+ : Array.isArray(last?.content)
125
+ ? ((last.content as Array<{ text?: string }>)
126
+ .map((b) => b?.text ?? "")
127
+ .join(""))
128
+ : "";
129
+ const isError =
130
+ last?.stopReason === "error" || typeof last?.errorMessage === "string";
131
+ if (!isError || lastText.length === 0) return currentTokens;
132
+ const parsed = parseWireTruth(lastText);
133
+ if (parsed == null) return currentTokens;
134
+ const wireTokens = parsed.requestTokens;
135
+ runtime.lastCtxTokens = wireTokens;
136
+ // EMA: the gap between the wire prompt and our message estimate (usage is
137
+ // absent in the 400 case, so estimateTokens is the true message-list estimate).
138
+ if (modelId !== "" && resolvedWindow > 0) {
139
+ const hSample = Math.max(0, wireTokens - estimateTokens);
140
+ sampleWireOverhead(modelId, runtime.currentStateDir, hSample, resolvedWindow);
141
+ }
142
+ // Prefer the provider's own available size for this event's math.
143
+ if (parsed.availableTokens > 0 &&
144
+ Math.abs(parsed.availableTokens - runtime.lastCtxWindow) > 0) {
145
+ runtime.lastCtxWindow = parsed.availableTokens;
146
+ }
147
+ runtime.diagCtxWireTruth++;
148
+ runtime.logger.info("wire_truth_parse", {
149
+ sessionId: runtime.rt.sessionId,
150
+ requestTokens: parsed.requestTokens,
151
+ availableTokens: parsed.availableTokens,
152
+ estimateTokens,
153
+ });
154
+ try {
155
+ runtime.appendEvent("wire_truth_parse", {
156
+ requestTokens: parsed.requestTokens,
157
+ availableTokens: parsed.availableTokens,
158
+ estimateTokens,
159
+ });
160
+ } catch {
161
+ /* non-fatal */
162
+ }
163
+ return wireTokens;
164
+ } catch {
165
+ /* non-fatal */
166
+ return currentTokens;
167
+ }
168
+ }
@@ -35,6 +35,12 @@ import {
35
35
  import { invokePipeline } from "./context-handler/pipelineRun.js";
36
36
  import { buildLiveTrimView } from "./context-handler/liveTrim.js";
37
37
  import { recapReplayedTail } from "./context-handler/headroom.js";
38
+ import {
39
+ sampleWireOverheadFromUsage,
40
+ applyWireTruthOverride,
41
+ resolveOverheadTokens,
42
+ correctEstimateWithOverhead,
43
+ } from "./context-handler/wireTruthApply.js";
38
44
 
39
45
  /** Register the context event handler (live-trim auto-trigger). */
40
46
  export function registerContextHandler(
@@ -99,15 +105,48 @@ export function registerContextHandler(
99
105
  // show empty/zero. Compute view lazily only when the fallback is needed
100
106
  // (at most one engineView call per context event; when auto is on and
101
107
  // usage.tokens is present, view is computed once below via reuse).
108
+ // v0.21.12: build the engineView whenever wireOverhead is ON (not only when
109
+ // usage is absent) so the EMA sampling + wire-truth blocks measure the REAL
110
+ // message-list estimate. Flag OFF keeps the v0.21.11 lazy path. Without
111
+ // this, estimateTokens falls back to the pct-derived value, which
112
+ // reconstructs usage.tokens exactly → hSample ≈ 0 → EMA trains to nothing.
102
113
  const viewForFallback =
103
- usage?.tokens == null ? runtime.engineView(messages) : null;
104
- const currentTokens =
105
- usage?.tokens ??
106
- (viewForFallback != null
114
+ usage?.tokens == null || config.wireOverhead ? runtime.engineView(messages) : null;
115
+ // v0.21.12: track WHICH source produced currentTokens so the invisible-
116
+ // overhead correction (H) is only applied to the ESTIMATE path. The
117
+ // provider-reported usage is ground truth for the message-list size; H is
118
+ // the gap between that and the wire prompt (system+tools+prepends).
119
+ const tokenSource: "usage" | "estimate" | "pct" =
120
+ usage?.tokens != null
121
+ ? "usage"
122
+ : viewForFallback != null
123
+ ? "estimate"
124
+ : "pct";
125
+ const estimateTokens =
126
+ viewForFallback != null
107
127
  ? estimateSessionTokens(viewForFallback)
108
- : null) ??
109
- Math.round(((pct ?? 0) / 100) * (usage?.contextWindow ?? 0));
128
+ : Math.round(((pct ?? 0) / 100) * (usage?.contextWindow ?? 0));
129
+ const rawTokens = usage?.tokens ?? estimateTokens;
130
+ // v0.21.12: invisible-overhead correction of the ESTIMATE-path count (see
131
+ // wireTruthApply.ts). modelId/resolvedWindow feed the EMA + tail-cap helpers.
132
+ const modelId = runtime.currentModel?.modelId ?? "";
133
+ const resolvedWindow =
134
+ usage?.contextWindow ?? (runtime.currentModel?.contextWindow ?? 0);
135
+ let currentTokens = correctEstimateWithOverhead({
136
+ config, tokenSource, rawTokens, modelId, resolvedWindow,
137
+ stateDir: runtime.currentStateDir,
138
+ });
110
139
  runtime.lastCtxTokens = currentTokens ?? null;
140
+ // v0.21.12: the invisible overhead H to feed the tail-cap budget at the
141
+ // fire/replay paths below (resolved once via the wireTruthApply helper so
142
+ // every call site passes the SAME H the gate used). 0 when the flag is OFF
143
+ // (byte-identical to v0.21.11) or no model/window is known.
144
+ const overheadTokens = resolveOverheadTokens({
145
+ config,
146
+ modelId,
147
+ resolvedWindow,
148
+ stateDir: runtime.currentStateDir,
149
+ });
111
150
  // 3WF-2: consume a pending live-window delta from a prior compaction. If a
112
151
  // compaction fired on the previous context event and the live window did
113
152
  // not shrink, this arms the ThrashGuard (meta). No-op when none pending.
@@ -132,6 +171,18 @@ export function registerContextHandler(
132
171
  reportedWindow > 0
133
172
  ? reportedWindow
134
173
  : (runtime.currentModel?.contextWindow ?? 0);
174
+ // v0.21.12: calibrate the invisible overhead H from EVERY context event that
175
+ // carries finite usage. estimateTokens is the REAL message-list estimate
176
+ // (engineView is built when wireOverhead is ON), so hSample = the true
177
+ // overhead (system+tools+prepends), not ≈0. Non-fatal; never throws.
178
+ sampleWireOverheadFromUsage({
179
+ runtime,
180
+ config,
181
+ modelId,
182
+ usageTokens: usage?.tokens,
183
+ resolvedWindow,
184
+ estimateTokens,
185
+ });
135
186
  runtime.snapshot(ctx);
136
187
  if (!config.auto) {
137
188
  const tailed = tailResult();
@@ -145,6 +196,22 @@ export function registerContextHandler(
145
196
  // message is captured, even if we don't compact this turn. Non-fatal.
146
197
  appendMirrorAndLedger(runtime, config, messages);
147
198
 
199
+ // v0.21.12: WIRED-TRUTH gate override (extracted to wireTruthApply.ts).
200
+ // When the last assistant message is an error whose text matches the
201
+ // provider's 400 shape, the parsed requestTokens become ground-truth
202
+ // currentTokens for THIS event's gate — breaking the 400 loop when the
203
+ // estimate path undercounts by ~50% (the v0.21.11 blind spot). No-op when
204
+ // wireOverhead is OFF (byte-identical).
205
+ currentTokens = applyWireTruthOverride({
206
+ runtime,
207
+ config,
208
+ messages,
209
+ modelId,
210
+ resolvedWindow,
211
+ estimateTokens,
212
+ currentTokens,
213
+ });
214
+
148
215
  // S29 FAST GATE: drive the auto-trigger off the context percent (see
149
216
  // gateCheck.ts). Returns a tailed view when the gate does not pass.
150
217
  const gate = evaluateGate(runtime, config, { pct, currentTokens, tailResult });
@@ -190,6 +257,7 @@ export function registerContextHandler(
190
257
  maxOutputTokens: runtime.currentModel?.maxTokens ?? 0,
191
258
  outputReservePct: config.outputReservePct,
192
259
  safetyMarginPct: runtime.trimCache.safetyMarginPct,
260
+ overheadTokens,
193
261
  });
194
262
  runtime.diagLiveTrimFires++; // trim view returned this call (replay counts as a fire)
195
263
  runtime.diagLiveTrimReplays++;
@@ -240,6 +308,7 @@ export function registerContextHandler(
240
308
  pct,
241
309
  currentTokens,
242
310
  tailResult,
311
+ overheadTokens,
243
312
  });
244
313
  if (pipeline.kind === "return") return pipeline.view;
245
314
 
@@ -321,6 +390,7 @@ export function registerContextHandler(
321
390
  ran: pipeline.ran,
322
391
  perModelThreshold: gate.perModelThreshold,
323
392
  tailResult,
393
+ overheadTokens,
324
394
  });
325
395
  });
326
396
  }
@@ -40,6 +40,8 @@ export class RuntimeInstrumentation {
40
40
  diagCtxThrown = 0; // live-trim try threw (caught)
41
41
  diagCtxOutputErrorTrip = 0; // Phase H: output-error catch tripped a forced compaction
42
42
  diagCtxHeadroomTrip = 0; // v0.21.9: output-headroom gate tripped a pre-overflow compaction
43
+ diagCtxWireTruth = 0; // v0.21.12: a provider 400 text was parsed into ground-truth tokens
44
+ diagCtxSkipCapped = 0; // v0.21.12: the live-trim skip path was tail-capped to fit the budget
43
45
 
44
46
  // Context health instrumentation (v0.12): rolling ring buffers for
45
47
  // drift detection + cache poison Layer 1 hash baseline.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-mega-compact",
3
- "version": "0.21.10",
3
+ "version": "0.21.12",
4
4
  "description": "Layered, local, vector-backed context compressor for pi — supersede/collapse/cluster compaction with deduped inline recall.",
5
5
  "type": "module",
6
6
  "license": "BSD-3-Clause",