pi-mega-compact 0.21.8 → 0.21.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/dedup/degenerate.js +68 -0
  2. package/dist/extensions/dashboard-server/routes-dedup-attribution.js +8 -1
  3. package/dist/extensions/dashboard-server/routes-rag-settings-compaction.js +42 -0
  4. package/dist/extensions/dashboard-server/routes-rag-settings-helpers.js +8 -8
  5. package/dist/extensions/mega-config.js +12 -0
  6. package/dist/extensions/mega-events/context-handler/gateCheck.js +51 -1
  7. package/dist/extensions/mega-events/context-handler/headroom.js +128 -0
  8. package/dist/extensions/mega-events/context-handler/liveTrim.js +37 -51
  9. package/dist/extensions/mega-events/context-handler/pipelineRun.js +14 -1
  10. package/dist/extensions/mega-events/context-handler.js +26 -3
  11. package/dist/extensions/mega-pipeline/compact/run.js +14 -3
  12. package/dist/extensions/mega-runtime/dashboard-snapshot.js +1 -0
  13. package/dist/extensions/mega-runtime/runtime-instrumentation.js +1 -0
  14. package/dist/extensions/mega-runtime/runtime-snapshot.js +1 -0
  15. package/dist/src/config/dedup.js +3 -0
  16. package/dist/src/dedup/degenerate.js +68 -0
  17. package/dist/src/extractive-salvage.js +195 -0
  18. package/dist/src/extractive.js +63 -72
  19. package/dist/src/vector-cortex/dedup-attr/rollup.js +5 -0
  20. package/dist/src/vectorStore/add-degenerate.js +25 -0
  21. package/dist/src/vectorStore/add.js +28 -5
  22. package/dist/src/vectorStore/dedup-audit.js +8 -0
  23. package/dist/vector-cortex/dedup-attr/rollup.js +5 -0
  24. package/dist/vectorStore/dedup-audit.js +8 -0
  25. package/extensions/dashboard-server/api-contracts/endpoints/types.ts +2 -0
  26. package/extensions/dashboard-server/routes-dedup-attribution.ts +11 -1
  27. package/extensions/dashboard-server/routes-rag-settings-compaction.ts +91 -0
  28. package/extensions/dashboard-server/routes-rag-settings-helpers.ts +13 -27
  29. package/extensions/mega-config-types.ts +25 -0
  30. package/extensions/mega-config.ts +12 -0
  31. package/extensions/mega-dashboard.ts +4 -1
  32. package/extensions/mega-events/context-handler/gateCheck.ts +67 -0
  33. package/extensions/mega-events/context-handler/headroom.ts +190 -0
  34. package/extensions/mega-events/context-handler/liveTrim.ts +37 -57
  35. package/extensions/mega-events/context-handler/pipelineRun.ts +14 -1
  36. package/extensions/mega-events/context-handler.ts +26 -3
  37. package/extensions/mega-pipeline/compact/run.ts +14 -4
  38. package/extensions/mega-runtime/dashboard-snapshot.ts +3 -0
  39. package/extensions/mega-runtime/runtime-instrumentation.ts +4 -0
  40. package/extensions/mega-runtime/runtime-snapshot.ts +2 -0
  41. package/package.json +1 -1
  42. package/src/config/dedup.ts +15 -0
  43. package/src/dedup/degenerate.ts +125 -0
  44. package/src/extractive-salvage.ts +212 -0
  45. package/src/extractive.ts +70 -75
  46. package/src/vector-cortex/dedup-attr/rollup.ts +4 -0
  47. package/src/vectorStore/add-degenerate.ts +64 -0
  48. package/src/vectorStore/add.ts +29 -5
  49. package/src/vectorStore/dedup-audit.ts +25 -2
@@ -0,0 +1,68 @@
1
+ /**
2
+ * The effective token floor for a checkpoint: the larger of the absolute floor
3
+ * and `MIN_PCT × originalTokenEstimate`.
4
+ *
5
+ * A missing / zero / non-finite `originalTokenEstimate` contributes nothing, so
6
+ * the absolute floor applies alone — direct add() callers and pre-v0.4 rows that
7
+ * never recorded the original region size are judged on absolute size only,
8
+ * never accidentally deemed degenerate by a 0-valued percentage term.
9
+ */
10
+ export function degenerateFloor(subject, tunables) {
11
+ const orig = subject.originalTokenEstimate;
12
+ const relative = typeof orig === "number" && Number.isFinite(orig) && orig > 0
13
+ ? orig * tunables.DEDUP_DEGEN_MIN_PCT
14
+ : 0;
15
+ return Math.max(tunables.DEDUP_DEGEN_MIN_TOKENS, relative);
16
+ }
17
+ /**
18
+ * Is this stored checkpoint a degenerate (content-free) summary?
19
+ *
20
+ * Calibration against the incident data:
21
+ * - skeleton: tokenEstimate 34, original ≈19166 → 34 < max(48, 95.8) → TRUE
22
+ * - normal: tokenEstimate 2000, original 70000 → 2000 > max(48, 350) → FALSE
23
+ *
24
+ * The relative term is what makes this scale: a 34-token summary of a 900-token
25
+ * region is a legitimate 26× compression, while the same 34 tokens standing in
26
+ * for 19k is a skeleton.
27
+ */
28
+ export function isDegenerateCheckpoint(subject, tunables) {
29
+ const tokens = subject.tokenEstimate ?? 0;
30
+ return tokens < degenerateFloor(subject, tunables);
31
+ }
32
+ /**
33
+ * Should an L1/L2 match be DECLINED because it would collapse richer incoming
34
+ * content onto a degenerate stored checkpoint?
35
+ *
36
+ * Returns true only when all four hold:
37
+ * 1. the umbrella flag is ON,
38
+ * 2. the matched (stored) checkpoint is degenerate,
39
+ * 3. the candidate is strictly richer than the match,
40
+ * 4. the candidate's content is not byte-identical to the match's.
41
+ *
42
+ * Condition 3 uses a strict `>`: equal-size skeletons collapsing is harmless and
43
+ * keeps the store from growing one row per compaction while the summarizer is
44
+ * broken. Only a genuine improvement is worth declining a collapse for.
45
+ *
46
+ * Condition 4 is a CORRECTNESS requirement, not a refinement. `context_chunks`
47
+ * carries a partial UNIQUE index on (session_id, content_hash) (schema/core.ts
48
+ * QA #1), so declining a match whose content hash already exists would fall
49
+ * through to an INSERT that throws — inside add(), which sits on the agent loop.
50
+ * It is also the semantically right call: identical bytes are the SAME region,
51
+ * so re-storing them adds no information and heals nothing. Only L0 may own the
52
+ * exact-match case; the guard exists for fuzzy matches on genuinely different
53
+ * text, which is exactly the incident's shape (each compaction produced a
54
+ * *similar but distinct* skeleton).
55
+ */
56
+ export function shouldSkipDegenerateMatch(matched, candidate, tunables) {
57
+ if (!tunables.DEDUP_DEGENERATE_GUARD)
58
+ return false;
59
+ if (!isDegenerateCheckpoint(matched, tunables))
60
+ return false;
61
+ if ((candidate.tokenEstimate ?? 0) <= (matched.tokenEstimate ?? 0))
62
+ return false;
63
+ // Byte-identical content → not a healing opportunity (and would violate the
64
+ // UNIQUE index). Compared only when both hashes are known.
65
+ const a = candidate.contentHash;
66
+ const b = matched.contentHash;
67
+ return !(a !== undefined && b !== undefined && a === b);
68
+ }
@@ -50,7 +50,14 @@ function parseAuditLine(line) {
50
50
  const status = obj.status;
51
51
  if (tier !== "L0" && tier !== "L1" && tier !== "L2" && tier !== "new")
52
52
  return null;
53
- if (status !== "deduped" && status !== "passed" && status !== "stored")
53
+ // "skipped" = a tier matched but the degenerate-match guard declined to
54
+ // collapse. Accepted so the line is not silently dropped from the tail; the
55
+ // rollup below counts only deduped/passed, so tier catch-share math is
56
+ // unchanged by its presence.
57
+ if (status !== "deduped" &&
58
+ status !== "passed" &&
59
+ status !== "stored" &&
60
+ status !== "skipped")
54
61
  return null;
55
62
  // sessionId is never read by the rollup; a parsed line may omit richer fields.
56
63
  return { type: "dedup_audit", ts: obj.ts, tier, status, sessionId: "" };
@@ -0,0 +1,42 @@
1
+ /**
2
+ * dashboard-server/routes-rag-settings-compaction.ts — Compaction SETTINGS.
3
+ *
4
+ * The compaction-group flag inventory, split out of routes-rag-settings-helpers.ts
5
+ * (delegate-shell split per the extensions/ 400-line soft limit) when the
6
+ * v0.21.9 output-headroom flags landed. regression_check.py globs every
7
+ * routes-rag-settings*.ts sibling, so no scanner update is needed on split.
8
+ *
9
+ * PREVENT-011: no `any` type.
10
+ */
11
+ const boolDirect = (key, label, description, def) => ({
12
+ key,
13
+ label,
14
+ description,
15
+ type: "boolean",
16
+ default: def,
17
+ disabledConvention: false,
18
+ requiresLlm: false,
19
+ });
20
+ const num = (key, label, description, def, min, max, unit) => ({
21
+ key,
22
+ label,
23
+ description,
24
+ type: "number",
25
+ default: def,
26
+ disabledConvention: false,
27
+ requiresLlm: false,
28
+ min,
29
+ max,
30
+ ...(unit ? { unit } : {}),
31
+ });
32
+ /** The compaction flags, as one SETTINGS category. */
33
+ export const COMPACTION_SETTINGS = {
34
+ name: "Compaction",
35
+ settings: [
36
+ num("MEGACOMPACT_THRESHOLD_PCT", "Compaction Threshold", "Fraction of the actual model context window at which compaction fires — 0.80 fires at 80% used (leaves 20% free). Applies to any model size; a per-model Model Thresholds row overrides it", 0.8, 0.1, 0.95),
37
+ num("MEGACOMPACT_THRASH_REARM_PCT", "Thrash Re-arm %", "After an ineffective compaction (live window did not shrink), refuse to re-fire until the live window grows by this fraction of the effective threshold. Default 0.10 (10%)", 0.1, 0.01, 0.5),
38
+ boolDirect("MEGACOMPACT_OUTPUT_ERROR_COMPACT", "Output-Error Compact", "When a model response is truncated mid-output (stopReason: 'length'), trip a one-shot forced compaction to free input headroom. Closes the small-context deadlock where the model truncates below the input threshold.", true),
39
+ boolDirect("MEGACOMPACT_OVERFLOW_HEADROOM", "Overflow Headroom Gate", "Fire compaction BEFORE the request overflows the model window — when input tokens + the output reserve + safety margin would exceed the context window — instead of waiting for the percent fire point (which judges only INPUT and never trips on small-window models whose output budget is a large fraction of the window). Percent-based: the reserve scales with the model's own window, so the math holds at every window size (32k…5M). OFF disables this pre-fire check (the gate reverts to input-only judgment); the pair-safe tail-cap hardenings are unconditional safety fixes and remain active.", true),
40
+ num("MEGACOMPACT_OUTPUT_RESERVE_PCT", "Output Reserve %", "FALLBACK output reserve as a fraction of the context window, used only when the model's declared maxTokens is absent or implausible (0, or a models.json sentinel like 1e9/1e38, or >= the window). When maxTokens IS plausible the declared value wins — vLLM-style backends reserve the FULL declared maxTokens. Default 0.30 (30%), clamped 0.10–0.95.", 0.3, 0.1, 0.95),
41
+ ],
42
+ };
@@ -1,4 +1,5 @@
1
1
  import { VECTOR_CORTEX_SETTINGS } from "./routes-rag-settings-vector-cortex.js";
2
+ import { COMPACTION_SETTINGS } from "./routes-rag-settings-compaction.js";
2
3
  // Shorthand builders to keep the inventory terse and unambiguous.
3
4
  const boolFlag = (key, label, description, def, requiresLlm = false) => ({
4
5
  key,
@@ -84,6 +85,7 @@ export const SETTINGS = [
84
85
  boolDirect("MEGACOMPACT_MARK_ONLY_L1", "Mark Only L1", "L1 runs but does not collapse", false),
85
86
  boolDirect("MEGACOMPACT_MARK_ONLY_L2", "Mark Only L2", "L2 runs but does not collapse", false),
86
87
  boolDirect("MEGACOMPACT_MINILM", "MiniLM Embedder", "Use MiniLM instead of trigram", false),
88
+ boolDirect("MEGACOMPACT_DEDUP_DEGENERATE_GUARD", "Degenerate Match Guard", "Decline an L1/L2 collapse when the MATCHED stored checkpoint is a content-free skeleton (a ~30-40 token structural summary) and the incoming region is richer. Without this, one degenerate checkpoint absorbs every later compaction forever and the store can never heal. OFF = byte-identical pre-guard cascade. Calibrated by the two Degenerate floors under Dedup Thresholds.", true),
87
89
  boolDirect("MEGACOMPACT_DEDUP_AUDIT", "Dedup Audit Trail", "Append one events.log line per tier decision (which layer collapsed a region, onto what, at what similarity) to tune the thresholds below. Pure instrumentation — dedup behavior is identical either way.", true),
88
90
  ],
89
91
  },
@@ -93,6 +95,8 @@ export const SETTINGS = [
93
95
  num("MEGACOMPACT_L2_THRESHOLD", "L2 Cosine Threshold", "L2 semantic dedup firing point", 0.85, 0, 1),
94
96
  num("MEGACOMPACT_L1_JACCARD", "L1 Jaccard Threshold", "L1 MinHash near-dup threshold", 0.8, 0, 1),
95
97
  num("MEGACOMPACT_DEDUP_SIM", "Dedup Similarity", "Legacy content-similarity fallback", 0.9, 0, 1),
98
+ num("MEGACOMPACT_DEDUP_DEGEN_MIN_TOKENS", "Degenerate Min Tokens", "Absolute token floor below which a stored summary counts as a degenerate skeleton (Degenerate Match Guard)", 48, 0, 10000, "tokens"),
99
+ num("MEGACOMPACT_DEDUP_DEGEN_MIN_PCT", "Degenerate Min Percent", "Relative floor as a fraction of the summary's original region size; a summary under max(min-tokens, pct x original) is degenerate", 0.005, 0, 1),
96
100
  num("MEGACOMPACT_RECALL_MIN_COSINE", "Recall Min Cosine (same-repo)", "3WF-3 same-repo floor the 3-source validator applies to the top winner (cross-repo 0.90 stays separate)", 0.12, 0, 1),
97
101
  num("MEGACOMPACT_MMR_LAMBDA", "MMR Lambda", "Maximal Marginal Relevance diversity", 0.5, 0, 1),
98
102
  num("MEGACOMPACT_SEMDEDUP_COSINE", "SemDeDup Cosine", "Offline SemDeDup pair threshold", 0.95, 0, 1),
@@ -127,14 +131,10 @@ export const SETTINGS = [
127
131
  num("MEGACOMPACT_EMBEDDING_CHARS_PER_TOKEN", "Embedding Chars per Token", "Estimated characters per token used for embedder chunking size", 4, 1, 32),
128
132
  ],
129
133
  },
130
- {
131
- name: "Compaction",
132
- settings: [
133
- num("MEGACOMPACT_THRESHOLD_PCT", "Compaction Threshold", "Fraction of the actual model context window at which compaction fires — 0.80 fires at 80% used (leaves 20% free). Applies to any model size; a per-model Model Thresholds row overrides it", 0.8, 0.1, 0.95),
134
- num("MEGACOMPACT_THRASH_REARM_PCT", "Thrash Re-arm %", "After an ineffective compaction (live window did not shrink), refuse to re-fire until the live window grows by this fraction of the effective threshold. Default 0.10 (10%)", 0.1, 0.01, 0.5),
135
- boolDirect("MEGACOMPACT_OUTPUT_ERROR_COMPACT", "Output-Error Compact", "When a model response is truncated mid-output (stopReason: 'length'), trip a one-shot forced compaction to free input headroom. Closes the small-context deadlock where the model truncates below the input threshold.", true),
136
- ],
137
- },
134
+ // v0.21.9: compaction group extracted to keep this file under the
135
+ // extensions/ soft limit (delegate-shell split); carries the overflow-
136
+ // headroom + output-reserve flags alongside the pre-existing trio.
137
+ COMPACTION_SETTINGS,
138
138
  {
139
139
  name: "Three-Way Failback",
140
140
  settings: [
@@ -210,6 +210,18 @@ export function loadConfig() {
210
210
  // Phase H: output-error catch — trip compaction on a truncated model output
211
211
  // (S28 stopReason==='length'). Default ON; OFF byte-identical pre-H.
212
212
  outputErrorCompact: envBool("MEGACOMPACT_OUTPUT_ERROR_COMPACT", true),
213
+ // v0.21.9 OUTPUT-HEADROOM GATE: fire compaction BEFORE the request
214
+ // overflows the model window (input + output reserve + margin >= window),
215
+ // not after. Percent-based: the reserve scales with the model's own window
216
+ // so the math holds at every window size (32k…5M). Default ON;
217
+ // OFF = byte-identical pre-v0.21.9 (2026-08-19 32k incident fix).
218
+ overflowHeadroom: envBool("MEGACOMPACT_OVERFLOW_HEADROOM", true),
219
+ // v0.21.9: fallback OUTPUT reserve as a FRACTION of the context window,
220
+ // used when the model's declared maxTokens is absent or implausible
221
+ // (0 / sentinel 1e9/1e38 / >= window). Clamped [0.1, 0.95]; default 0.30.
222
+ // When maxTokens IS plausible the declared value wins (vLLM reserves the
223
+ // full maxTokens) — this fraction is only the fallback.
224
+ outputReservePct: clamp(envFlag("MEGACOMPACT_OUTPUT_RESERVE_PCT", 0.3), 0.1, 0.95),
213
225
  windowDedupe: envBool("MEGACOMPACT_WINDOW_DEDUPE", true),
214
226
  recallTailInject: envBool("MEGACOMPACT_RECALL_TAIL_INJECT", true),
215
227
  // 3WF-1: TriggerGuard — guarantee a staged recall block on every context
@@ -1,6 +1,7 @@
1
1
  import { resolveModelThreshold, DEFAULT_SAFETY_MARGIN_PCT, DEFAULT_FIRE_POINT_PCT, } from "../../../src/store/sqlite.js";
2
2
  import { autoCompactCheck } from "../../../src/compact.js";
3
3
  import { isThrashBlockedFor } from "./thrashGuard.js";
4
+ import { resolveOutputReserve } from "./headroom.js";
4
5
  /**
5
6
  * 3WF-2 ThrashGuard consult — refuse to fire a NEW compaction while the guard
6
7
  * is armed. After an ineffective compaction (the live window did not shrink),
@@ -20,11 +21,18 @@ import { isThrashBlockedFor } from "./thrashGuard.js";
20
21
  * Umbrella OFF ⇒ always false (byte-identical to v0.20.83). Non-fatal: a store
21
22
  * read error returns false — never refuse compaction on a store fault.
22
23
  */
23
- export function thrashGuardBlocks(runtime, config, currentTokens) {
24
+ export function thrashGuardBlocks(runtime, config, currentTokens, headroomExceeded) {
24
25
  if (!config.threeWayFailback)
25
26
  return false;
26
27
  if (currentTokens == null)
27
28
  return false;
29
+ // v0.21.9: an overflow-bound fire (headroomExceeded) is EXEMPT from the
30
+ // thrash guard. The guard exists to stop wasted re-compaction when the
31
+ // window refuses to shrink; but an overflowed request is not "wasted work"
32
+ // — it is the model about to 400. Blocking that fire reproduces the
33
+ // 2026-08-19 32k deadlock (compact never → request > window → error loop).
34
+ if (headroomExceeded)
35
+ return false;
28
36
  return isThrashBlockedFor(runtime, currentTokens, runtime.currentStateDir);
29
37
  }
30
38
  /**
@@ -67,6 +75,48 @@ export function evaluateGate(runtime, config, opts) {
67
75
  runtime.diagCtxOutputErrorTrip++;
68
76
  return { kind: "proceed", perModelThreshold };
69
77
  }
78
+ // v0.21.9 OUTPUT-HEADROOM GATE (the root-cause fix for the 32k truncation
79
+ // loop). The percent/token fire points above judge only INPUT utilization
80
+ // (tier% of the window), but a request's budget is
81
+ // input tokens + the model's output budget + safety margin.
82
+ // On a small-window model with a large maxTokens (the user's 32k/20k
83
+ // GLM-4.7), the request overflows at ~32% INPUT (21.4k + 20k > 32.768k) —
84
+ // long before any percent gate fires → provider 400 "request exceeds the
85
+ // available context size" every turn → the poisoned-error loop. Phase H only
86
+ // reacts to stopReason 'length' (mid-output truncation); a pre-output 400
87
+ // never arms it, so "compact never". This check fires the compaction
88
+ // BEFORE the overflow instead of after.
89
+ //
90
+ // PERCENT-BASED (LTS invariant — must work at every window size: 32k, 64k,
91
+ // 200k, 1M, 5M): the reserve is a FRACTION of the model's own window via
92
+ // resolveOutputReserve (plausible declared maxTokens wins, else
93
+ // clamp(MEGACOMPACT_OUTPUT_RESERVE_PCT, 10–95%) × window). Same math, any
94
+ // size. window <= 0 (unknown) ⇒ deferred (never guess a window), matching
95
+ // the effectiveThresholdImpl Phase-C invariant. Gated on
96
+ // config.overflowHeadroom (default ON; OFF = byte-identical pre-v0.21.9).
97
+ // Thrash-guard exemption: headroomExceeded rides along on the proceed so the
98
+ // handler's thrash consult never refuses an overflow-bound fire (see
99
+ // thrashGuardBlocks above) — an overflowed session is unrecoverable, so a
100
+ // wasted re-fire is always the better outcome (2026-08-19 incident).
101
+ if (config.overflowHeadroom &&
102
+ runtime.lastCtxWindow > 0 &&
103
+ Number.isFinite(currentTokens) &&
104
+ currentTokens > 0) {
105
+ const { reserveTokens, fallbackUsed } = resolveOutputReserve(runtime.lastCtxWindow, runtime.currentModel?.maxTokens ?? 0, config.outputReservePct);
106
+ const headroomMargin = Math.ceil(runtime.lastCtxWindow * (perModelThreshold.safetyMarginPct / 100));
107
+ if (currentTokens + reserveTokens + headroomMargin >= runtime.lastCtxWindow) {
108
+ runtime.diagCtxHeadroomTrip++;
109
+ runtime.logger.info("gate-headroom-trip", {
110
+ sessionId: runtime.rt.sessionId,
111
+ currentTokens,
112
+ ctxWindow: runtime.lastCtxWindow,
113
+ reserveTokens,
114
+ fallbackUsed,
115
+ marginPct: perModelThreshold.safetyMarginPct,
116
+ });
117
+ return { kind: "proceed", perModelThreshold, headroomExceeded: true };
118
+ }
119
+ }
70
120
  // S29 FAST GATE: `custom` (absolute MEGACOMPACT_THRESHOLD_TOKENS,
71
121
  // tierPct null) is an explicit opt-out of percent scaling — it keeps the
72
122
  // token gate. When pct is unavailable (window unknown / a model that
@@ -0,0 +1,128 @@
1
+ import { estimateBlockTokens, estimateMessageTokens } from "../../../src/tokens.js";
2
+ import { messageContentText } from "./messageText.js";
3
+ /**
4
+ * The model's declared maxTokens is only trusted as the output budget when it
5
+ * is plausible. models.json carries sentinel junk for some entries (1e9,
6
+ * 1e38, "unlimited"), and some providers report 0/absent. A declared budget
7
+ * above this FRACTION of the window is implausible — fall back to the
8
+ * configured fraction so a 200k/1e9 model doesn't compute a negative budget
9
+ * and silently disable the cap (the pre-v0.21.9 bug). Percent-based: holds at
10
+ * every window size.
11
+ *
12
+ * WHY 0.95 AND NOT LOWER: vLLM-style backends reject a request when
13
+ * `input + max_tokens > context window` — they reserve the model's FULL
14
+ * declared maxTokens, not a fraction of it. The user's own GLM-4.7 entry is
15
+ * 32000/20000 (62.5%); a 0.6 cutoff rejected that REAL config as
16
+ * "implausible" and fell back to a 30% reserve (9600) while the backend
17
+ * reserved the full 20000 — the gate would keep firing late and the
18
+ * post-compact tail would still overflow (2026-08-19 incident, attempt #6).
19
+ * A declared budget is plausible up to just below the WHOLE window; anything
20
+ * at/above the window (or the 1e9/1e38 sentinels) is junk.
21
+ */
22
+ export const MAX_OUTPUT_PLAUSIBLE_FRACTION = 0.95;
23
+ /** Bounds for the fallback reserve fraction (MEGACOMPACT_OUTPUT_RESERVE_PCT). */
24
+ export const OUTPUT_RESERVE_PCT_MIN = 0.1;
25
+ export const OUTPUT_RESERVE_PCT_MAX = 0.95;
26
+ /**
27
+ * Resolve the output reserve (tokens) for a model window.
28
+ *
29
+ * - window <= 0 (unknown) → { reserveTokens: 0, fallbackUsed: false }; every
30
+ * consumer is guarded on window > 0 and defers (never guesses a window).
31
+ * - maxTokens plausible (0 < maxTokens <= 95% of the window) → maxTokens —
32
+ * vLLM-style backends reserve the FULL declared maxTokens, so the reserve
33
+ * must equal it, not a fraction of it.
34
+ * - otherwise → clamp(outputReservePct, 0.1, 0.95) × window.
35
+ *
36
+ * `outputReservePct` is config.outputReservePct (already env-clamped at load,
37
+ * re-clamped here for defense against direct callers).
38
+ */
39
+ export function resolveOutputReserve(ctxWindow, maxTokens, outputReservePct) {
40
+ if (!Number.isFinite(ctxWindow) || ctxWindow <= 0) {
41
+ return { reserveTokens: 0, fallbackUsed: false };
42
+ }
43
+ const plausible = Number.isFinite(maxTokens) &&
44
+ maxTokens > 0 &&
45
+ maxTokens <= ctxWindow * MAX_OUTPUT_PLAUSIBLE_FRACTION;
46
+ if (plausible)
47
+ return { reserveTokens: Math.round(maxTokens), fallbackUsed: false };
48
+ const pct = Math.min(OUTPUT_RESERVE_PCT_MAX, Math.max(OUTPUT_RESERVE_PCT_MIN, Number.isFinite(outputReservePct) ? outputReservePct : 0.3));
49
+ return { reserveTokens: Math.ceil(ctxWindow * pct), fallbackUsed: true };
50
+ }
51
+ /**
52
+ * Pair-safe front-drop for the live-trim tail cap. Drops OLDEST messages from
53
+ * the front of `recentRaw` until the remaining tail fits
54
+ * `ctxWindow − outputReserve − safetyMargin − summaryTokens`, then advances
55
+ * the start index past any leading toolResult messages so the preserved tail
56
+ * never begins on an orphaned toolResult (PREVENT-PI-002: a toolCall/toolResult
57
+ * pair must not be split). Never returns an empty tail — the final message is
58
+ * always kept so the agent can respond.
59
+ *
60
+ * v0.21.9 hardenings over the pre-v0.21.9 inline cap in liveTrim.ts:
61
+ * 1. BUDGET FLOOR — when the reserve exceeds the window (implausible maxTokens
62
+ * made budget <= 0) the old block silently skipped the cap entirely and an
63
+ * oversized tail sailed past the window. Now the reserve is clamped (via
64
+ * resolveOutputReserve) to a fraction of the window, so a floor budget
65
+ * always exists. If even ONE message exceeds the floor budget we keep only
66
+ * the final message — the agent's last turn is the one thing the model
67
+ * must always see.
68
+ * 2. TOOL-PAIR SAFETY — the old front-drop could land between a toolCall and
69
+ * its toolResult, splitting the pair.
70
+ *
71
+ * Pure: returns { recent, dropped } without touching the input array.
72
+ */
73
+ export function applyTailCap(opts) {
74
+ const { recentRaw, summaryTokens, ctxWindow, outputReservePct } = opts;
75
+ if (ctxWindow <= 0 || recentRaw.length <= 1) {
76
+ return { recent: [...recentRaw], dropped: 0 };
77
+ }
78
+ const msgTokens = opts.messageTokens && opts.messageTokens.length === recentRaw.length
79
+ ? opts.messageTokens
80
+ : null;
81
+ const { reserveTokens } = resolveOutputReserve(ctxWindow, opts.maxOutputTokens, outputReservePct);
82
+ const safetyMargin = Math.ceil(ctxWindow * (Math.max(0, opts.safetyMarginPct) / 100));
83
+ // Budget floor: never negative. An implausible reserve (clamped above to
84
+ // <= 95% of the window) plus margin + summary can still exceed the window
85
+ // on tiny summaries-free edges; the floor keeps the cap alive with a small
86
+ // positive budget instead of disabling it (pre-v0.21.9 behavior).
87
+ const budget = Math.max(1, ctxWindow - reserveTokens - safetyMargin - Math.max(0, summaryTokens));
88
+ let start = 0;
89
+ let tailTokens = 0;
90
+ for (let i = recentRaw.length - 1; i >= 0; i--) {
91
+ tailTokens +=
92
+ msgTokens != null
93
+ ? Math.max(0, msgTokens[i])
94
+ : estimateMessageTokens({ text: messageContentText(recentRaw[i]) });
95
+ if (tailTokens > budget) {
96
+ // Keep from i+1 onward; never drop below the FINAL message.
97
+ start = Math.min(i + 1, recentRaw.length - 1);
98
+ break;
99
+ }
100
+ }
101
+ // PREVENT-PI-002: never begin the preserved tail on an orphaned toolResult —
102
+ // its toolCall was dropped by the front-cut above. Advance past consecutive
103
+ // toolResults; the pair stays intact or drops whole.
104
+ while (start < recentRaw.length - 1 &&
105
+ recentRaw[start].role === "toolResult") {
106
+ start++;
107
+ }
108
+ return { recent: recentRaw.slice(start), dropped: start };
109
+ }
110
+ /**
111
+ * v0.21.9: re-cap a REPLAYED trim tail (D.2 in context-handler.ts, D.3 in
112
+ * pipelineRun.ts). The replay paths return the cached trim view verbatim —
113
+ * which bypasses the fire-time tail cap. A model switch mid-epoch can shrink
114
+ * the window, leaving a replayed tail that fit the OLD window overflowing the
115
+ * NEW one. Re-runs applyTailCap against the CURRENT window with the margin
116
+ * stored at fire time (trimCache.safetyMarginPct), so the replayed view never
117
+ * exceeds what the gate would allow. Pure — no runtime dependency.
118
+ */
119
+ export function recapReplayedTail(opts) {
120
+ return applyTailCap({
121
+ recentRaw: opts.recentRaw,
122
+ summaryTokens: estimateBlockTokens(messageContentText(opts.summaryAgentMsg)),
123
+ ctxWindow: opts.ctxWindow,
124
+ maxOutputTokens: opts.maxOutputTokens,
125
+ outputReservePct: opts.outputReservePct,
126
+ safetyMarginPct: opts.safetyMarginPct,
127
+ });
128
+ }
@@ -1,6 +1,6 @@
1
- import { estimateBlockTokens, estimateMessageTokens, } from "../../../src/tokens.js";
1
+ import { estimateBlockTokens } from "../../../src/tokens.js";
2
2
  import { computeLiveTrimCut, liveTrimSummaryMessage } from "../../mega-trim.js";
3
- import { messageContentText } from "./messageText.js";
3
+ import { applyTailCap } from "./headroom.js";
4
4
  /**
5
5
  * Reconstruct the live-trim window (summary + recent anchor) for this LLM
6
6
  * call. Returns the tailed view, or undefined when no trim is safe this call.
@@ -79,56 +79,33 @@ export function buildLiveTrimView(runtime, config, ctx, opts) {
79
79
  // but has NO token cap, so a 2-message tail of two 80K bash outputs sails
80
80
  // right past the window.
81
81
  //
82
- // Cap: when the model context window is known, reserve room for the
83
- // summary + the model's max output tokens + a 10% safety margin, then
84
- // drop oldest preserved messages from the front of `recentRaw` until the
85
- // tail fits. Never drops below the FINAL message (always keep the latest
86
- // turn so the agent can respond). This is a last-resort HARD cap it
87
- // only fires when the preserved tail alone is oversized, which is rare.
82
+ // v0.21.9: the reserve + front-drop now lives in headroom.ts (single
83
+ // source shared with the gate's pre-fire headroom check and the D.2/D.3
84
+ // replay paths): (a) percent-based reserve plausible declared maxTokens
85
+ // wins, else clamp(MEGACOMPACT_OUTPUT_RESERVE_PCT, 10–95%) × window so
86
+ // the math is identical at any window size and a sentinel maxTokens
87
+ // (1e9/1e38) can no longer drive the budget negative and silently
88
+ // disable the cap; (b) budget floor (max(1, …)) so the cap stays active
89
+ // on every window; (c) pair-safe front-drop — the preserved tail never
90
+ // begins on an orphaned toolResult (PREVENT-PI-002).
88
91
  const ctxWindow = runtime.lastCtxWindow;
89
92
  // Reuse the per-model threshold resolved at the gate (single lookup).
90
93
  const modelThreshold = perModelThreshold;
91
- // Reserve room for output tokens. Use the model's reported max output
92
- // when known; fall back to 10% of the window (scales with any model —
93
- // 20K for a 200K window, 100K for a 1M window) so we never let the
94
- // preserved tail eat the model's output budget when maxTokens is unknown.
95
- const maxOutput = runtime.currentModel?.maxTokens && runtime.currentModel.maxTokens > 0
96
- ? runtime.currentModel.maxTokens
97
- : Math.ceil(ctxWindow * 0.1);
98
- let recent = recentRaw;
99
- if (ctxWindow > 0 && recentRaw.length > 1) {
100
- const summaryTokens = estimateBlockTokens(summaryMsg.text);
101
- // Reserve: summary + max output + per-model safety margin (0-20%).
102
- const safetyMargin = Math.ceil(ctxWindow * (modelThreshold.safetyMarginPct / 100));
103
- const budget = ctxWindow - maxOutput - safetyMargin - summaryTokens;
104
- if (budget > 0) {
105
- // Walk recent from the front, dropping oldest first until the
106
- // remaining tail fits. Use the AgentMessage→engine-text estimate via
107
- // messageContentText (already imported) + estimateMessageTokens.
108
- let tailTokens = 0;
109
- for (let i = recentRaw.length - 1; i >= 0; i--) {
110
- const m = recentRaw[i];
111
- tailTokens += estimateMessageTokens({
112
- text: messageContentText(m),
113
- });
114
- if (tailTokens > budget) {
115
- // Keep from i+1 onward; but never fewer than the final message.
116
- const startIdx = Math.min(i + 1, recentRaw.length - 1);
117
- if (startIdx > 0) {
118
- recent = recentRaw.slice(startIdx);
119
- runtime.logger.warn("live-trim-tail-cap", {
120
- sessionId: runtime.rt.sessionId,
121
- dropped: startIdx,
122
- tailTokens,
123
- safetyMarginPct: modelThreshold.safetyMarginPct,
124
- budget,
125
- ctxWindow,
126
- });
127
- }
128
- break;
129
- }
130
- }
131
- }
94
+ const { recent, dropped } = applyTailCap({
95
+ recentRaw,
96
+ summaryTokens: estimateBlockTokens(summaryMsg.text),
97
+ ctxWindow,
98
+ maxOutputTokens: runtime.currentModel?.maxTokens ?? 0,
99
+ outputReservePct: config.outputReservePct,
100
+ safetyMarginPct: modelThreshold.safetyMarginPct,
101
+ });
102
+ if (dropped > 0) {
103
+ runtime.logger.warn("live-trim-tail-cap", {
104
+ sessionId: runtime.rt.sessionId,
105
+ dropped,
106
+ safetyMarginPct: modelThreshold.safetyMarginPct,
107
+ ctxWindow,
108
+ });
132
109
  }
133
110
  // v0.8.6: cache the trim view so subsequent gated calls in this epoch
134
111
  // replay it verbatim (stabilizing the KV-cache prefix) instead of
@@ -138,8 +115,12 @@ export function buildLiveTrimView(runtime, config, ctx, opts) {
138
115
  // (rt.lastCheckpointId) instead of ran.result.checkpointId, which is
139
116
  // dedup-volatile: on a re-compact that dedups onto a DIFFERENT existing
140
117
  // checkpoint, result.checkpointId is the matched id (engine.ts:188) while
141
- // lastCheckpointId is only updated on a genuinely new checkpoint
142
- // (compact.ts:100-104). Keying on result.checkpointId would make
118
+ // lastCheckpointId was, pre-C1, only updated on a genuinely new checkpoint.
119
+ // C1 (v0.21.10) now stamps lastCheckpointId on the dedup path too (see
120
+ // compact/run.ts) — it means "the checkpoint backing this epoch" — so this
121
+ // key and the D.2/D.3 comparison agree in both directions and the `??`
122
+ // fallbacks below are now only for the truly-no-checkpoint edge case.
123
+ // Keying on result.checkpointId directly would still make
143
124
  // trimCache.checkpointId != rt.lastCheckpointId forever after that
144
125
  // dedup fire, disabling replay for the rest of the epoch (the
145
126
  // alternating cache-miss that 0.8.6 meant to fix). Prefer the stable
@@ -152,6 +133,11 @@ export function buildLiveTrimView(runtime, config, ctx, opts) {
152
133
  summaryAgentMsg,
153
134
  ctxPct: pct ?? null,
154
135
  ctxTokens: currentTokens,
136
+ // v0.21.9: the D.2/D.3 replay paths re-cap the replayed tail against
137
+ // the CURRENT window (a model switch can change it mid-epoch). The
138
+ // margin % used at fire time is stored alongside so the replay uses
139
+ // the same reserve math as the fire that built the view.
140
+ safetyMarginPct: modelThreshold.safetyMarginPct,
155
141
  };
156
142
  runtime.snapshot(ctx);
157
143
  // DIAG (team-run relief): confirm the live trim actually fires + how big
@@ -2,6 +2,7 @@ import { runCompact } from "../../mega-pipeline.js";
2
2
  import { pressureFromPct, pressureRatio } from "../../mega-config.js";
3
3
  import { recordCompactLatency } from "../../mega-runtime/vc-observer.js";
4
4
  import { decideLivePath } from "../../mega-runtime/vector-cortex-live.js";
5
+ import { recapReplayedTail } from "./headroom.js";
5
6
  import { defaultClock } from "../../../src/vector-cortex/rollout/gate.js";
6
7
  import { VC5C_ENABLED } from "../../../src/config/vector-cortex.js";
7
8
  /**
@@ -72,7 +73,19 @@ export function invokePipeline(pi, runtime, config, ctx, opts) {
72
73
  if (runtime.trimCache &&
73
74
  runtime.trimCache.checkpointId === runtime.rt.lastCheckpointId &&
74
75
  runtime.trimCache.cut <= opts.messages.length) {
75
- const recent = opts.messages.slice(runtime.trimCache.cut); // guardrails-allow PREVENT-PI-002: cached `cut` was sanitized by computeLiveTrimCut (src/boundary.ts); replayed verbatim, transcript only grows within an epoch.
76
+ const recentRaw = opts.messages.slice(runtime.trimCache.cut); // guardrails-allow PREVENT-PI-002: cached `cut` was sanitized by computeLiveTrimCut (src/boundary.ts); replayed verbatim, transcript only grows within an epoch.
77
+ // v0.21.9: RE-CAP the replayed tail against the CURRENT window —
78
+ // the D.3 skip-replay bypasses the fire-time tail cap exactly like
79
+ // D.2; a model switch mid-epoch can shrink the window below what
80
+ // the cached view was built for. No-op when the tail already fits.
81
+ const { recent } = recapReplayedTail({
82
+ recentRaw,
83
+ summaryAgentMsg: runtime.trimCache.summaryAgentMsg,
84
+ ctxWindow: runtime.lastCtxWindow,
85
+ maxOutputTokens: runtime.currentModel?.maxTokens ?? 0,
86
+ outputReservePct: config.outputReservePct,
87
+ safetyMarginPct: runtime.trimCache.safetyMarginPct,
88
+ });
76
89
  runtime.diagLiveTrimFires++;
77
90
  runtime.diagLiveTrimReplays++;
78
91
  runtime.snapshot(ctx);
@@ -9,6 +9,7 @@ import { evaluateGate, thrashGuardBlocks } from "./context-handler/gateCheck.js"
9
9
  import { markCompactionFired, evaluatePendingReduction, } from "./context-handler/thrashGuard.js";
10
10
  import { invokePipeline } from "./context-handler/pipelineRun.js";
11
11
  import { buildLiveTrimView } from "./context-handler/liveTrim.js";
12
+ import { recapReplayedTail } from "./context-handler/headroom.js";
12
13
  /** Register the context event handler (live-trim auto-trigger). */
13
14
  export function registerContextHandler(pi, runtime, config) {
14
15
  // ---- Auto-trigger: live trim (compact and continue) + native durable ----
@@ -139,7 +140,23 @@ export function registerContextHandler(pi, runtime, config) {
139
140
  : currentTokens - (runtime.trimCache.ctxTokens ?? 0) >=
140
141
  runtime.effectiveThreshold * 0.5;
141
142
  if (!grewEnough) {
142
- const recent = messages.slice(runtime.trimCache.cut); // guardrails-allow PREVENT-PI-002: cached `cut` was sanitized once by computeLiveTrimCut (src/boundary.ts) and replayed verbatim; the transcript only grows within an epoch (cache is cleared on durable truncation), so the preserved run still starts on a toolPair-safe index.
143
+ const recentRaw = messages.slice(runtime.trimCache.cut); // guardrails-allow PREVENT-PI-002: cached `cut` was sanitized once by computeLiveTrimCut (src/boundary.ts) and replayed verbatim; the transcript only grows within an epoch (cache is cleared on durable truncation), so the preserved run still starts on a toolPair-safe index.
144
+ // v0.21.9: RE-CAP the replayed tail against the CURRENT window.
145
+ // Replay returns the cached view verbatim, which bypasses the
146
+ // fire-time tail cap — a model switch mid-epoch can shrink the
147
+ // window and leave a replayed tail that fit the OLD window
148
+ // overflowing the NEW one. Same reserve math as the fire
149
+ // (margin % stored in the cache at fire time); no-op when the
150
+ // tail already fits. Pair-safe (applyTailCap advances past any
151
+ // leading toolResult its front-drop exposes).
152
+ const { recent } = recapReplayedTail({
153
+ recentRaw,
154
+ summaryAgentMsg: runtime.trimCache.summaryAgentMsg,
155
+ ctxWindow: runtime.lastCtxWindow,
156
+ maxOutputTokens: runtime.currentModel?.maxTokens ?? 0,
157
+ outputReservePct: config.outputReservePct,
158
+ safetyMarginPct: runtime.trimCache.safetyMarginPct,
159
+ });
143
160
  runtime.diagLiveTrimFires++; // trim view returned this call (replay counts as a fire)
144
161
  runtime.diagLiveTrimReplays++;
145
162
  runtime.snapshot(ctx);
@@ -159,15 +176,21 @@ export function registerContextHandler(pi, runtime, config) {
159
176
  // invokePipeline (the real fire point), so it covers the percent + token
160
177
  // gate paths alike. Umbrella OFF ⇒ never blocks (byte-identical). Returns
161
178
  // the tailed view so a staged recall block still rides along.
162
- if (thrashGuardBlocks(runtime, config, currentTokens)) {
179
+ if (thrashGuardBlocks(runtime, config, currentTokens, gate.headroomExceeded)) {
163
180
  runtime.diagCtxFastGate++;
164
181
  runtime.snapshot(ctx);
165
182
  return tailResult() ?? undefined;
166
183
  }
167
184
  // Debounce so we don't fire on every context event past threshold.
168
185
  // (Replay already returned above — only fresh compacts reach this point.)
186
+ // C2 (v0.21.10): EXEMPT headroom-triggered fires, matching the thrash-guard
187
+ // exemption above. pi's own overflow recovery (400 → compact → immediate
188
+ // retry) re-fires a context event <2s after our last fire; debouncing it
189
+ // returned the RAW untrimmed view, so input + output reserve still blew the
190
+ // window → 400 → "recovery failed after one compact-and-retry attempt".
191
+ // An overflowed session is unrecoverable; a re-fire is merely wasteful.
169
192
  const now = Date.now();
170
- if (now < runtime.debounceUntil) {
193
+ if (now < runtime.debounceUntil && !gate.headroomExceeded) {
171
194
  runtime.diagCtxDebounce++;
172
195
  return tailResult() ?? undefined;
173
196
  }
@@ -71,10 +71,21 @@ function doCompact(view, keepFrom, opts, sid, config, pi, ctx, runtime) {
71
71
  runtime.pulsing = false;
72
72
  if (result.skipped)
73
73
  return { skipped: true };
74
- if (!result.deduped) {
74
+ // C1 (v0.21.10): lastCheckpointId tracks "the checkpoint backing this epoch",
75
+ // so it is stamped on BOTH paths — a matched-dedup checkpoint backs this epoch
76
+ // just as much as a freshly created one. Previously the dedup path left it
77
+ // undefined, so a runtime session whose every compaction deduped (common after
78
+ // a process restart, when checkpoints persist but `rt` is rebuilt) never set it
79
+ // → liveTrim's trimCache fell back to result.checkpointId (the matched id) →
80
+ // `trimCache.checkpointId === rt.lastCheckpointId` was `"chkpt_001" !== undefined`
81
+ // → the D.2/D.3 replay NEVER matched and the full pipeline re-ran on every
82
+ // context event (liveTrimReplays: 0, "comp lag warn"). A later fire matching a
83
+ // DIFFERENT checkpoint now changes the key once (one cache regeneration), then
84
+ // replays stabilise. `persistedThisSession` keeps its narrower meaning ("we
85
+ // wrote NEW state this session") and stays gated on !deduped.
86
+ if (!result.deduped)
75
87
  runtime.rt.persistedThisSession = true;
76
- runtime.rt.lastCheckpointId = result.checkpointId;
77
- }
88
+ runtime.rt.lastCheckpointId = result.checkpointId;
78
89
  runtime.rt.lastCompactedFrom = result.compactedFrom;
79
90
  runtime.rt.lastCompactedTokens = result.tokenEstimate;
80
91
  runtime.rt.dedupAttempts++;