pi-mega-compact 0.21.9 → 0.21.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/dedup/degenerate.js +68 -0
- package/dist/extensions/dashboard-server/routes-dedup-attribution.js +8 -1
- package/dist/extensions/dashboard-server/routes-rag-settings-helpers.js +3 -0
- package/dist/extensions/mega-events/context-handler/headroom.js +58 -3
- package/dist/extensions/mega-events/context-handler/liveTrim.js +6 -2
- package/dist/extensions/mega-events/context-handler.js +7 -1
- package/dist/extensions/mega-pipeline/compact/run.js +14 -3
- package/dist/src/config/dedup.js +3 -0
- package/dist/src/dedup/degenerate.js +68 -0
- package/dist/src/extractive-salvage.js +195 -0
- package/dist/src/extractive.js +63 -72
- package/dist/src/vector-cortex/dedup-attr/rollup.js +5 -0
- package/dist/src/vectorStore/add-degenerate.js +25 -0
- package/dist/src/vectorStore/add.js +28 -5
- package/dist/src/vectorStore/dedup-audit.js +8 -0
- package/dist/vector-cortex/dedup-attr/rollup.js +5 -0
- package/dist/vectorStore/dedup-audit.js +8 -0
- package/extensions/dashboard-server/routes-dedup-attribution.ts +11 -1
- package/extensions/dashboard-server/routes-rag-settings-helpers.ts +8 -0
- package/extensions/mega-events/context-handler/headroom.ts +50 -3
- package/extensions/mega-events/context-handler/liveTrim.ts +6 -2
- package/extensions/mega-events/context-handler.ts +7 -1
- package/extensions/mega-pipeline/compact/run.ts +14 -4
- package/package.json +1 -1
- package/src/config/dedup.ts +15 -0
- package/src/dedup/degenerate.ts +125 -0
- package/src/extractive-salvage.ts +212 -0
- package/src/extractive.ts +70 -75
- package/src/vector-cortex/dedup-attr/rollup.ts +4 -0
- package/src/vectorStore/add-degenerate.ts +64 -0
- package/src/vectorStore/add.ts +29 -5
- package/src/vectorStore/dedup-audit.ts +25 -2
package/dist/src/extractive.js
CHANGED
|
@@ -8,12 +8,14 @@
|
|
|
8
8
|
* Deterministic: same messages → same output, every time.
|
|
9
9
|
*/
|
|
10
10
|
import { estimateBlockTokens } from "./tokens.js";
|
|
11
|
+
import { CURRENT_WORK_PATH_RE, isInterestingPath, isPlaceholderRequest, isSkeletonSummary, buildSalvageDigest, collectKeyFiles, extractFilesModified, } from "./extractive-salvage.js";
|
|
11
12
|
// ---- Limits ----------------------------------------------------------------
|
|
12
13
|
const MAX_RECENT_USER = 3;
|
|
13
14
|
const MAX_DECISIONS = 5;
|
|
14
|
-
const MAX_FILES = 10;
|
|
15
15
|
const MAX_PENDING = 5;
|
|
16
16
|
const MAX_TOPIC_LINES = 12;
|
|
17
|
+
/** Cap for the merged keyFiles ∪ filesModified "Key files" line (A2a). */
|
|
18
|
+
const MAX_SUMMARY_FILES = 8;
|
|
17
19
|
// ---- Truncation helper -----------------------------------------------------
|
|
18
20
|
function truncate(s, maxLen) {
|
|
19
21
|
if (s.length <= maxLen)
|
|
@@ -27,8 +29,22 @@ function truncate(s, maxLen) {
|
|
|
27
29
|
* Captures: tools used, recent user requests, current work, key files,
|
|
28
30
|
* pending work. Typically 12 lines / ~500 tokens instead of ~70K.
|
|
29
31
|
*/
|
|
30
|
-
function buildTopicSummary(messages, tools, recentUser, currentWork, keyFiles, pending) {
|
|
32
|
+
function buildTopicSummary(messages, tools, recentUser, currentWork, keyFiles, pending, filesModified, decisions) {
|
|
31
33
|
const lines = [];
|
|
34
|
+
// A2a: files captured from write/edit tool inputs are extracted
|
|
35
|
+
// extension-agnostically but never reached the summary. Fold them in so work
|
|
36
|
+
// outside the recency window (and outside the path regex) is still reported.
|
|
37
|
+
// Drop an absolute path when a kept relative path already names the SAME file.
|
|
38
|
+
// Only MULTI-COMPONENT relative paths fold ("engine/mesh.go" absorbs
|
|
39
|
+
// "/proj/engine/mesh.go"); a bare basename ("mesh.go") is never folded into an
|
|
40
|
+
// absolute path, since "/proj/other/x/mesh.go" may be a genuinely different
|
|
41
|
+
// file (QA lens 1 finding: the naive endsWith dropped different directories).
|
|
42
|
+
const combined = [...keyFiles, ...filesModified];
|
|
43
|
+
const allFiles = [...new Set(combined)].filter((p) => {
|
|
44
|
+
if (!p.startsWith("/"))
|
|
45
|
+
return true;
|
|
46
|
+
return !combined.some((r) => r !== p && !r.startsWith("/") && r.includes("/") && p.endsWith("/" + r));
|
|
47
|
+
}).slice(0, MAX_SUMMARY_FILES);
|
|
32
48
|
// Scope line
|
|
33
49
|
const users = messages.filter((m) => m.role === "user");
|
|
34
50
|
const assistants = messages.filter((m) => m.role === "assistant");
|
|
@@ -45,45 +61,59 @@ function buildTopicSummary(messages, tools, recentUser, currentWork, keyFiles, p
|
|
|
45
61
|
// Current work
|
|
46
62
|
if (currentWork)
|
|
47
63
|
lines.push(`Current work: ${currentWork}`);
|
|
48
|
-
// Key files
|
|
49
|
-
if (
|
|
50
|
-
lines.push(`Key files: ${
|
|
64
|
+
// Key files (keyFiles ∪ filesModified)
|
|
65
|
+
if (allFiles.length)
|
|
66
|
+
lines.push(`Key files: ${allFiles.join(", ")}.`);
|
|
51
67
|
// Pending work
|
|
52
68
|
if (pending.length) {
|
|
53
69
|
lines.push("Pending work:");
|
|
54
70
|
for (const p of pending)
|
|
55
71
|
lines.push(` • ${p}`);
|
|
56
72
|
}
|
|
73
|
+
// A2c: a scope-line-only summary carries zero information and strands a
|
|
74
|
+
// resumed session. Salvage the tail of the conversation instead. The line cap
|
|
75
|
+
// is raised ONLY here: the salvage block is bounded at 5 lines + 1 header, and
|
|
76
|
+
// a skeleton by definition contributed just the 1 scope line, so the worst
|
|
77
|
+
// case is 7 lines — still well under the normal 12-line budget.
|
|
78
|
+
const skeleton = isSkeletonSummary({ recentUser, keyFiles: allFiles, currentWork, decisions, pending });
|
|
79
|
+
if (skeleton) {
|
|
80
|
+
const digest = buildSalvageDigest(messages);
|
|
81
|
+
if (digest.length) {
|
|
82
|
+
lines.push("Recent activity:");
|
|
83
|
+
for (const d of digest)
|
|
84
|
+
lines.push(` • ${d}`);
|
|
85
|
+
}
|
|
86
|
+
return lines.join("\n");
|
|
87
|
+
}
|
|
57
88
|
// Cap total length
|
|
58
89
|
return lines.slice(0, MAX_TOPIC_LINES).join("\n");
|
|
59
90
|
}
|
|
60
|
-
// ----
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
const ext = m[2];
|
|
68
|
-
const basename = filePath.split("/").pop() ?? filePath;
|
|
69
|
-
if (basename === "node_modules" || filePath.includes("node_modules/"))
|
|
70
|
-
continue;
|
|
71
|
-
if (INTERESTING_EXT.has(ext))
|
|
72
|
-
paths.push(filePath);
|
|
73
|
-
}
|
|
74
|
-
return paths;
|
|
75
|
-
}
|
|
76
|
-
// ---- Recent user requests (existing logic, kept) ---------------------------
|
|
91
|
+
// ---- Recent user requests --------------------------------------------------
|
|
92
|
+
/**
|
|
93
|
+
* A2b: skip content-free "resume"/"continue" turns and look further back to
|
|
94
|
+
* fill the quota, so a resumed session surfaces its real requests. Falls back
|
|
95
|
+
* to the placeholders when EVERY user turn is one (an honest "• resume" beats
|
|
96
|
+
* an empty section).
|
|
97
|
+
*/
|
|
77
98
|
function collectRecentUserRequests(messages, limit) {
|
|
78
|
-
const
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
99
|
+
const substantive = [];
|
|
100
|
+
const placeholders = [];
|
|
101
|
+
for (let i = messages.length - 1; i >= 0 && substantive.length < limit; i--) {
|
|
102
|
+
if (messages[i].role !== "user")
|
|
103
|
+
continue;
|
|
104
|
+
let snippet = messages[i].text.split("\n").slice(0, 3).join(" ");
|
|
105
|
+
snippet = snippet.replace(/^.+\nProcessed\$?\s*/i, "").replace(/\n/g, " ");
|
|
106
|
+
const cleaned = truncate(snippet, 200);
|
|
107
|
+
if (!cleaned.trim())
|
|
108
|
+
continue;
|
|
109
|
+
if (isPlaceholderRequest(cleaned)) {
|
|
110
|
+
if (placeholders.length < limit)
|
|
111
|
+
placeholders.push(cleaned);
|
|
112
|
+
continue;
|
|
84
113
|
}
|
|
114
|
+
substantive.push(cleaned);
|
|
85
115
|
}
|
|
86
|
-
return
|
|
116
|
+
return (substantive.length ? substantive : placeholders).reverse();
|
|
87
117
|
}
|
|
88
118
|
// ---- Pending work (existing logic, kept) -----------------------------------
|
|
89
119
|
const PENDING_WORDS = ["todo", "next", "pending", "follow up", "remaining"];
|
|
@@ -106,8 +136,9 @@ function inferCurrentWork(messages) {
|
|
|
106
136
|
const m = messages[i];
|
|
107
137
|
if (m.role !== "assistant")
|
|
108
138
|
continue;
|
|
109
|
-
|
|
110
|
-
|
|
139
|
+
// A1: same language-agnostic policy as extractFilePaths.
|
|
140
|
+
const path = m.text.match(CURRENT_WORK_PATH_RE);
|
|
141
|
+
if (path && isInterestingPath(path[1], path[2])) {
|
|
111
142
|
const line = m.text.split("\n").slice(0, 2).join(" ");
|
|
112
143
|
return truncate(line, 200);
|
|
113
144
|
}
|
|
@@ -144,30 +175,6 @@ function extractDecisions(messages) {
|
|
|
144
175
|
}
|
|
145
176
|
return [...new Set(decisions)];
|
|
146
177
|
}
|
|
147
|
-
// ---- Files modified --------------------------------------------------------
|
|
148
|
-
function extractFilesModified(tools) {
|
|
149
|
-
const files = new Set();
|
|
150
|
-
for (const m of tools) {
|
|
151
|
-
if (!m.toolName)
|
|
152
|
-
continue;
|
|
153
|
-
const name = m.toolName.toLowerCase();
|
|
154
|
-
if (name === "write" || name === "edit" || name === "notebookedit") {
|
|
155
|
-
// Extract file path from input payload
|
|
156
|
-
const input = m.input ?? m.text;
|
|
157
|
-
const pathMatch = input.match(/["']?(\/[^\s"']+\.\w+)["']?/);
|
|
158
|
-
if (pathMatch)
|
|
159
|
-
files.add(pathMatch[1]);
|
|
160
|
-
}
|
|
161
|
-
if (name === "bash") {
|
|
162
|
-
const cmd = m.input ?? m.text;
|
|
163
|
-
if (cmd.includes("git add") || cmd.includes("git commit") || cmd.includes("git diff")) {
|
|
164
|
-
for (const p of extractFilePaths(cmd))
|
|
165
|
-
files.add(p);
|
|
166
|
-
}
|
|
167
|
-
}
|
|
168
|
-
}
|
|
169
|
-
return [...files].slice(0, MAX_FILES);
|
|
170
|
-
}
|
|
171
178
|
// ---- Public API ------------------------------------------------------------
|
|
172
179
|
/**
|
|
173
180
|
* Deterministic extractive summary. Same messages → same output, every time.
|
|
@@ -192,23 +199,7 @@ export function extractiveSummarize(messages) {
|
|
|
192
199
|
const pending = inferPendingWork(safe);
|
|
193
200
|
const keyDecisions = extractDecisions(safe);
|
|
194
201
|
const filesModified = extractFilesModified(toolMsgs);
|
|
195
|
-
const topicSummary = buildTopicSummary(safe, tools, recentUser, currentWork, keyFiles, pending);
|
|
202
|
+
const topicSummary = buildTopicSummary(safe, tools, recentUser, currentWork, keyFiles, pending, filesModified, keyDecisions);
|
|
196
203
|
const tokenEstimate = estimateBlockTokens(topicSummary);
|
|
197
204
|
return { topicSummary, keyDecisions, nextSteps: pending, filesModified, tokenEstimate };
|
|
198
205
|
}
|
|
199
|
-
// ---- Key files (existing logic from compact.ts, moved here) ----------------
|
|
200
|
-
const MAX_KEY_FILES = 5;
|
|
201
|
-
const FRESHNESS_WINDOW = 10;
|
|
202
|
-
function collectKeyFiles(messages) {
|
|
203
|
-
const recent = messages.slice(-FRESHNESS_WINDOW);
|
|
204
|
-
const pathFreq = new Map();
|
|
205
|
-
for (const m of recent) {
|
|
206
|
-
for (const p of extractFilePaths(m.text)) {
|
|
207
|
-
pathFreq.set(p, (pathFreq.get(p) ?? 0) + 1);
|
|
208
|
-
}
|
|
209
|
-
}
|
|
210
|
-
return [...pathFreq.entries()]
|
|
211
|
-
.sort((a, b) => b[1] - a[1])
|
|
212
|
-
.slice(0, MAX_KEY_FILES)
|
|
213
|
-
.map(([p]) => p);
|
|
214
|
-
}
|
|
@@ -49,6 +49,11 @@ export function computeDedupTierRollup(events, windowMs, now) {
|
|
|
49
49
|
const ts = Date.parse(ev.ts);
|
|
50
50
|
if (Number.isNaN(ts) || ts < windowStart || ts > windowEnd)
|
|
51
51
|
continue;
|
|
52
|
+
// "skipped" (degenerate-match guard declined a collapse) is a decision, but
|
|
53
|
+
// NOT a tier catch — it is attributed to no tier below. Excluding it from the
|
|
54
|
+
// denominator keeps l0Share + l1Share + l2Share summing to 1 over the window.
|
|
55
|
+
if (ev.status === "skipped")
|
|
56
|
+
continue;
|
|
52
57
|
total += 1;
|
|
53
58
|
switch (ev.tier) {
|
|
54
59
|
case "L0":
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { shouldSkipDegenerateMatch } from "../dedup/degenerate.js";
|
|
2
|
+
/**
|
|
3
|
+
* Build the decliner for one add() cascade.
|
|
4
|
+
*
|
|
5
|
+
* Returns a predicate the L1/L2 call sites use as a one-line guard. When it
|
|
6
|
+
* returns true it has ALREADY recorded the declined decision (monitoring event +
|
|
7
|
+
* `skipped` audit line + the live `onTier` detail), so the caller only has to
|
|
8
|
+
* fall through. When the umbrella flag is off it always returns false and emits
|
|
9
|
+
* nothing — byte-identical to the pre-guard cascade.
|
|
10
|
+
*/
|
|
11
|
+
export function degenerateDecliner(args) {
|
|
12
|
+
const { store, input, contentHash, cfg, audit, t0 } = args;
|
|
13
|
+
const candidate = { ...input, contentHash };
|
|
14
|
+
return (tier, matched, similarity) => {
|
|
15
|
+
if (!shouldSkipDegenerateMatch(matched, candidate, cfg))
|
|
16
|
+
return false;
|
|
17
|
+
// Reported as `mark_only`: a tier matched but policy declined to collapse —
|
|
18
|
+
// exactly the existing MARK_ONLY shape, with a distinct reason string so the
|
|
19
|
+
// dashboard can tell a guard decline from an operator-configured MARK_ONLY.
|
|
20
|
+
store.record(tier, "mark_only", "degenerateGuard", Date.now() - t0, similarity, matched.checkpointId);
|
|
21
|
+
audit.skipped(tier, matched.checkpointId, "degenerateGuard", similarity);
|
|
22
|
+
input.onTier?.({ tier, status: "passed", detail: "degenerateGuard" });
|
|
23
|
+
return true;
|
|
24
|
+
};
|
|
25
|
+
}
|
|
@@ -27,6 +27,7 @@ import { lshBands } from "../dedup/l1-lsh.js";
|
|
|
27
27
|
import { openBloom, saveBloom } from "../store/bloom.js";
|
|
28
28
|
import { listCheckpoints, nextCheckpointId, upsertCheckpoint, loadSessionState, saveSessionState, upsertMinhashSignature, insertLshBuckets, addTokensSaved, bumpDedupStats, } from "../store/sqlite.js";
|
|
29
29
|
import { computeRegionHash } from "./hash.js";
|
|
30
|
+
import { degenerateDecliner } from "./add-degenerate.js";
|
|
30
31
|
import { runL0Tier, computeSummaryHash } from "./add-l0.js";
|
|
31
32
|
import { findL1Duplicate } from "./add-l1.js";
|
|
32
33
|
import { dedupAuditRecorder } from "./dedup-audit.js";
|
|
@@ -67,8 +68,22 @@ export function addCheckpoint(store, input) {
|
|
|
67
68
|
// Tracks whether a tier matched while in MARK_ONLY (record-but-don't-collapse),
|
|
68
69
|
// and which tier.
|
|
69
70
|
let markOnly = null;
|
|
70
|
-
//
|
|
71
|
+
// Content digest for this candidate (also consumed by the L0 tier below).
|
|
71
72
|
const digest = computeContentDigest(input.regionText);
|
|
73
|
+
// Degenerate-match guard (incident 2026-08-19): declines a fuzzy-tier collapse
|
|
74
|
+
// onto a content-free skeleton when the incoming region is richer, so the
|
|
75
|
+
// skeleton stops absorbing every future compaction. Returns true (having
|
|
76
|
+
// already recorded the decision) ⇒ the caller treats the match as a non-match.
|
|
77
|
+
// Flag-off ⇒ always false ⇒ byte-identical predecessor. See add-degenerate.ts.
|
|
78
|
+
const declineDegenerate = degenerateDecliner({
|
|
79
|
+
store,
|
|
80
|
+
input,
|
|
81
|
+
contentHash: digest.contentHash,
|
|
82
|
+
cfg,
|
|
83
|
+
audit,
|
|
84
|
+
t0,
|
|
85
|
+
});
|
|
86
|
+
// L0 exact-match tier (contentHash / regionHash / summaryHash) — see add-l0.ts.
|
|
72
87
|
const bloom = openBloom(store.stateDir);
|
|
73
88
|
const summaryHash = computeSummaryHash(input.topicSummary);
|
|
74
89
|
const l0 = runL0Tier({
|
|
@@ -94,7 +109,10 @@ export function addCheckpoint(store, input) {
|
|
|
94
109
|
onTier?.({ tier: "L1", status: "scanning" });
|
|
95
110
|
if (cfg.L1_ENABLED) {
|
|
96
111
|
const l1 = findL1Duplicate(store, sessionId, input.regionText, all);
|
|
97
|
-
|
|
112
|
+
// Guard first: a declined match must not collapse and must not be recorded
|
|
113
|
+
// as a MARK_ONLY hit either — it is a non-match for the rest of the cascade.
|
|
114
|
+
const l1Declined = l1 !== undefined && declineDegenerate("L1", l1);
|
|
115
|
+
if (l1 && !l1Declined && !cfg.MARK_ONLY_L1) {
|
|
98
116
|
l1.timestamp = input.timestamp;
|
|
99
117
|
upsertCheckpoint(l1, store.stateDir);
|
|
100
118
|
bumpDedupStats(true, store.stateDir);
|
|
@@ -105,7 +123,7 @@ export function addCheckpoint(store, input) {
|
|
|
105
123
|
onTier?.({ tier: "L1", status: "deduped", detail: "l1MinHash" });
|
|
106
124
|
return r;
|
|
107
125
|
}
|
|
108
|
-
if (l1 && cfg.MARK_ONLY_L1)
|
|
126
|
+
if (l1 && !l1Declined && cfg.MARK_ONLY_L1)
|
|
109
127
|
markOnly = "L1";
|
|
110
128
|
}
|
|
111
129
|
onTier?.({ tier: "L1", status: "passed" });
|
|
@@ -131,7 +149,12 @@ export function addCheckpoint(store, input) {
|
|
|
131
149
|
const sim = cosineSimilarity(embedding, cp.embedding);
|
|
132
150
|
return sim > best.sim ? { checkpoint: cp, sim } : best;
|
|
133
151
|
}, { checkpoint: all[0], sim: -1 });
|
|
134
|
-
|
|
152
|
+
// A declined match falls through to the "store a fresh checkpoint" path; the
|
|
153
|
+
// guard already audited the decision, so the near-miss emit below is skipped.
|
|
154
|
+
const l2Declined = !timedOut &&
|
|
155
|
+
nearest.sim >= simThreshold &&
|
|
156
|
+
declineDegenerate("L2", nearest.checkpoint, nearest.sim);
|
|
157
|
+
if (!timedOut && !l2Declined && nearest.sim >= simThreshold) {
|
|
135
158
|
if (!cfg.MARK_ONLY_L2) {
|
|
136
159
|
// Near-identical — update timestamp on existing checkpoint
|
|
137
160
|
nearest.checkpoint.timestamp = input.timestamp;
|
|
@@ -164,7 +187,7 @@ export function addCheckpoint(store, input) {
|
|
|
164
187
|
// Near-miss: how close did we come to collapsing? Only emitted when the
|
|
165
188
|
// scan actually completed and scored a candidate — a timed-out scan has no
|
|
166
189
|
// honest best to report.
|
|
167
|
-
if (!timedOut && nearest.sim >= 0) {
|
|
190
|
+
if (!timedOut && !l2Declined && nearest.sim >= 0) {
|
|
168
191
|
audit.passed("L2", nearest.checkpoint.checkpointId, nearest.sim);
|
|
169
192
|
}
|
|
170
193
|
}
|
|
@@ -69,6 +69,14 @@ export function dedupAuditRecorder(ctx, scope) {
|
|
|
69
69
|
matchedEntry,
|
|
70
70
|
similarity,
|
|
71
71
|
}),
|
|
72
|
+
skipped: (tier, matchedEntry, dedupReason, similarity) => emitDedupAudit(ctx, {
|
|
73
|
+
...base,
|
|
74
|
+
tier,
|
|
75
|
+
status: "skipped",
|
|
76
|
+
matchedEntry,
|
|
77
|
+
dedupReason,
|
|
78
|
+
...(similarity === undefined ? {} : { similarity }),
|
|
79
|
+
}),
|
|
72
80
|
stored: (storedEntry, dedupReason, tokenEstimate) => emitDedupAudit(ctx, {
|
|
73
81
|
...base,
|
|
74
82
|
tier: "new",
|
|
@@ -49,6 +49,11 @@ export function computeDedupTierRollup(events, windowMs, now) {
|
|
|
49
49
|
const ts = Date.parse(ev.ts);
|
|
50
50
|
if (Number.isNaN(ts) || ts < windowStart || ts > windowEnd)
|
|
51
51
|
continue;
|
|
52
|
+
// "skipped" (degenerate-match guard declined a collapse) is a decision, but
|
|
53
|
+
// NOT a tier catch — it is attributed to no tier below. Excluding it from the
|
|
54
|
+
// denominator keeps l0Share + l1Share + l2Share summing to 1 over the window.
|
|
55
|
+
if (ev.status === "skipped")
|
|
56
|
+
continue;
|
|
52
57
|
total += 1;
|
|
53
58
|
switch (ev.tier) {
|
|
54
59
|
case "L0":
|
|
@@ -69,6 +69,14 @@ export function dedupAuditRecorder(ctx, scope) {
|
|
|
69
69
|
matchedEntry,
|
|
70
70
|
similarity,
|
|
71
71
|
}),
|
|
72
|
+
skipped: (tier, matchedEntry, dedupReason, similarity) => emitDedupAudit(ctx, {
|
|
73
|
+
...base,
|
|
74
|
+
tier,
|
|
75
|
+
status: "skipped",
|
|
76
|
+
matchedEntry,
|
|
77
|
+
dedupReason,
|
|
78
|
+
...(similarity === undefined ? {} : { similarity }),
|
|
79
|
+
}),
|
|
72
80
|
stored: (storedEntry, dedupReason, tokenEstimate) => emitDedupAudit(ctx, {
|
|
73
81
|
...base,
|
|
74
82
|
tier: "new",
|
|
@@ -59,7 +59,17 @@ function parseAuditLine(line: string): DedupAuditEvent | null {
|
|
|
59
59
|
const tier = obj.tier;
|
|
60
60
|
const status = obj.status;
|
|
61
61
|
if (tier !== "L0" && tier !== "L1" && tier !== "L2" && tier !== "new") return null;
|
|
62
|
-
|
|
62
|
+
// "skipped" = a tier matched but the degenerate-match guard declined to
|
|
63
|
+
// collapse. Accepted so the line is not silently dropped from the tail; the
|
|
64
|
+
// rollup below counts only deduped/passed, so tier catch-share math is
|
|
65
|
+
// unchanged by its presence.
|
|
66
|
+
if (
|
|
67
|
+
status !== "deduped" &&
|
|
68
|
+
status !== "passed" &&
|
|
69
|
+
status !== "stored" &&
|
|
70
|
+
status !== "skipped"
|
|
71
|
+
)
|
|
72
|
+
return null;
|
|
63
73
|
// sessionId is never read by the rollup; a parsed line may omit richer fields.
|
|
64
74
|
return { type: "dedup_audit", ts: obj.ts, tier, status, sessionId: "" };
|
|
65
75
|
}
|
|
@@ -228,6 +228,12 @@ export const SETTINGS: ReadonlyArray<SettingGroup> = [
|
|
|
228
228
|
boolDirect("MEGACOMPACT_MARK_ONLY_L1", "Mark Only L1", "L1 runs but does not collapse", false),
|
|
229
229
|
boolDirect("MEGACOMPACT_MARK_ONLY_L2", "Mark Only L2", "L2 runs but does not collapse", false),
|
|
230
230
|
boolDirect("MEGACOMPACT_MINILM", "MiniLM Embedder", "Use MiniLM instead of trigram", false),
|
|
231
|
+
boolDirect(
|
|
232
|
+
"MEGACOMPACT_DEDUP_DEGENERATE_GUARD",
|
|
233
|
+
"Degenerate Match Guard",
|
|
234
|
+
"Decline an L1/L2 collapse when the MATCHED stored checkpoint is a content-free skeleton (a ~30-40 token structural summary) and the incoming region is richer. Without this, one degenerate checkpoint absorbs every later compaction forever and the store can never heal. OFF = byte-identical pre-guard cascade. Calibrated by the two Degenerate floors under Dedup Thresholds.",
|
|
235
|
+
true,
|
|
236
|
+
),
|
|
231
237
|
boolDirect(
|
|
232
238
|
"MEGACOMPACT_DEDUP_AUDIT",
|
|
233
239
|
"Dedup Audit Trail",
|
|
@@ -242,6 +248,8 @@ export const SETTINGS: ReadonlyArray<SettingGroup> = [
|
|
|
242
248
|
num("MEGACOMPACT_L2_THRESHOLD", "L2 Cosine Threshold", "L2 semantic dedup firing point", 0.85, 0, 1),
|
|
243
249
|
num("MEGACOMPACT_L1_JACCARD", "L1 Jaccard Threshold", "L1 MinHash near-dup threshold", 0.8, 0, 1),
|
|
244
250
|
num("MEGACOMPACT_DEDUP_SIM", "Dedup Similarity", "Legacy content-similarity fallback", 0.9, 0, 1),
|
|
251
|
+
num("MEGACOMPACT_DEDUP_DEGEN_MIN_TOKENS", "Degenerate Min Tokens", "Absolute token floor below which a stored summary counts as a degenerate skeleton (Degenerate Match Guard)", 48, 0, 10000, "tokens"),
|
|
252
|
+
num("MEGACOMPACT_DEDUP_DEGEN_MIN_PCT", "Degenerate Min Percent", "Relative floor as a fraction of the summary's original region size; a summary under max(min-tokens, pct x original) is degenerate", 0.005, 0, 1),
|
|
245
253
|
num("MEGACOMPACT_RECALL_MIN_COSINE", "Recall Min Cosine (same-repo)", "3WF-3 same-repo floor the 3-source validator applies to the top winner (cross-repo 0.90 stays separate)", 0.12, 0, 1),
|
|
246
254
|
num("MEGACOMPACT_MMR_LAMBDA", "MMR Lambda", "Maximal Marginal Relevance diversity", 0.5, 0, 1),
|
|
247
255
|
num("MEGACOMPACT_SEMDEDUP_COSINE", "SemDeDup Cosine", "Offline SemDeDup pair threshold", 0.95, 0, 1),
|
|
@@ -16,9 +16,56 @@
|
|
|
16
16
|
* Pure functions, no runtime dependency — trivially unit-testable headlessly.
|
|
17
17
|
*/
|
|
18
18
|
import type { AgentMessage } from "@earendil-works/pi-agent-core";
|
|
19
|
-
import { estimateBlockTokens
|
|
19
|
+
import { estimateBlockTokens } from "../../../src/tokens.js";
|
|
20
20
|
import { messageContentText } from "./messageText.js";
|
|
21
21
|
|
|
22
|
+
/**
|
|
23
|
+
* Full-surface AgentMessage token estimate for BUDGET arithmetic (tail cap).
|
|
24
|
+
*
|
|
25
|
+
* convertToLlm (pi dist/core/messages.js) ships assistant/toolResult messages
|
|
26
|
+
* VERBATIM — every content block goes over the wire: text, thinking, toolCall
|
|
27
|
+
* (name + full `arguments` JSON), toolResult output, role wrappers. The text
|
|
28
|
+
* extractor (messageContentText) is lossy-on-purpose for analytics, and using
|
|
29
|
+
* it here made a GLM-4.7-style assistant message with ~11.6k bytes of toolCall
|
|
30
|
+
* arguments register as ~77 tokens — a 30k-token tail passed an 11.9k budget,
|
|
31
|
+
* the model overflowed, and pi's one-shot compact-and-retry failed
|
|
32
|
+
* ("Context overflow recovery failed", 2026-08-20 incident).
|
|
33
|
+
*
|
|
34
|
+
* Counts every byte the provider actually receives. Still a heuristic (len/4
|
|
35
|
+
* + 1 per block, like estimateBlockTokens) — just no longer lossy. Never
|
|
36
|
+
* throws: unknown block shapes fall back to their JSON serialization length,
|
|
37
|
+
* and a non-array/string content is counted as its serialization.
|
|
38
|
+
*/
|
|
39
|
+
export function estimateAgentMessageBudgetTokens(m: AgentMessage): number {
|
|
40
|
+
try {
|
|
41
|
+
const c = (m as { content?: unknown }).content;
|
|
42
|
+
let bytes = 0;
|
|
43
|
+
if (typeof c === "string") {
|
|
44
|
+
bytes += c.length;
|
|
45
|
+
} else if (Array.isArray(c)) {
|
|
46
|
+
for (const b of c) {
|
|
47
|
+
if (b == null || typeof b !== "object") continue;
|
|
48
|
+
const o = b as Record<string, unknown>;
|
|
49
|
+
if (typeof o.text === "string") bytes += o.text.length;
|
|
50
|
+
if (typeof o.thinking === "string") bytes += o.thinking.length;
|
|
51
|
+
if (typeof o.name === "string") bytes += o.name.length;
|
|
52
|
+
if (o.arguments != null) bytes += JSON.stringify(o.arguments).length;
|
|
53
|
+
if (typeof o.output === "string") bytes += o.output.length;
|
|
54
|
+
// Per-block envelope overhead (role/type markers), matching the
|
|
55
|
+
// len/4+1 block accounting in estimateBlockTokens.
|
|
56
|
+
bytes += 4;
|
|
57
|
+
}
|
|
58
|
+
} else if (c != null) {
|
|
59
|
+
bytes += JSON.stringify(c).length;
|
|
60
|
+
}
|
|
61
|
+
return estimateBlockTokens(" ".repeat(Math.max(0, bytes)));
|
|
62
|
+
} catch {
|
|
63
|
+
// non-fatal: fall back to the legacy text-only estimate rather than
|
|
64
|
+
// disable the cap on a pathological message.
|
|
65
|
+
return estimateBlockTokens(messageContentText(m));
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
22
69
|
/**
|
|
23
70
|
* The model's declared maxTokens is only trusted as the output budget when it
|
|
24
71
|
* is plausible. models.json carries sentinel junk for some entries (1e9,
|
|
@@ -143,7 +190,7 @@ export function applyTailCap(opts: {
|
|
|
143
190
|
tailTokens +=
|
|
144
191
|
msgTokens != null
|
|
145
192
|
? Math.max(0, msgTokens[i])
|
|
146
|
-
:
|
|
193
|
+
: estimateAgentMessageBudgetTokens(recentRaw[i]);
|
|
147
194
|
if (tailTokens > budget) {
|
|
148
195
|
// Keep from i+1 onward; never drop below the FINAL message.
|
|
149
196
|
start = Math.min(i + 1, recentRaw.length - 1);
|
|
@@ -181,7 +228,7 @@ export function recapReplayedTail(opts: {
|
|
|
181
228
|
}): { recent: AgentMessage[]; dropped: number } {
|
|
182
229
|
return applyTailCap({
|
|
183
230
|
recentRaw: opts.recentRaw,
|
|
184
|
-
summaryTokens:
|
|
231
|
+
summaryTokens: estimateAgentMessageBudgetTokens(opts.summaryAgentMsg),
|
|
185
232
|
ctxWindow: opts.ctxWindow,
|
|
186
233
|
maxOutputTokens: opts.maxOutputTokens,
|
|
187
234
|
outputReservePct: opts.outputReservePct,
|
|
@@ -170,8 +170,12 @@ export function buildLiveTrimView(
|
|
|
170
170
|
// (rt.lastCheckpointId) instead of ran.result.checkpointId, which is
|
|
171
171
|
// dedup-volatile: on a re-compact that dedups onto a DIFFERENT existing
|
|
172
172
|
// checkpoint, result.checkpointId is the matched id (engine.ts:188) while
|
|
173
|
-
// lastCheckpointId
|
|
174
|
-
// (
|
|
173
|
+
// lastCheckpointId was, pre-C1, only updated on a genuinely new checkpoint.
|
|
174
|
+
// C1 (v0.21.10) now stamps lastCheckpointId on the dedup path too (see
|
|
175
|
+
// compact/run.ts) — it means "the checkpoint backing this epoch" — so this
|
|
176
|
+
// key and the D.2/D.3 comparison agree in both directions and the `??`
|
|
177
|
+
// fallbacks below are now only for the truly-no-checkpoint edge case.
|
|
178
|
+
// Keying on result.checkpointId directly would still make
|
|
175
179
|
// trimCache.checkpointId != rt.lastCheckpointId forever after that
|
|
176
180
|
// dedup fire, disabling replay for the rest of the epoch (the
|
|
177
181
|
// alternating cache-miss that 0.8.6 meant to fix). Prefer the stable
|
|
@@ -219,8 +219,14 @@ export function registerContextHandler(
|
|
|
219
219
|
|
|
220
220
|
// Debounce so we don't fire on every context event past threshold.
|
|
221
221
|
// (Replay already returned above — only fresh compacts reach this point.)
|
|
222
|
+
// C2 (v0.21.10): EXEMPT headroom-triggered fires, matching the thrash-guard
|
|
223
|
+
// exemption above. pi's own overflow recovery (400 → compact → immediate
|
|
224
|
+
// retry) re-fires a context event <2s after our last fire; debouncing it
|
|
225
|
+
// returned the RAW untrimmed view, so input + output reserve still blew the
|
|
226
|
+
// window → 400 → "recovery failed after one compact-and-retry attempt".
|
|
227
|
+
// An overflowed session is unrecoverable; a re-fire is merely wasteful.
|
|
222
228
|
const now = Date.now();
|
|
223
|
-
if (now < runtime.debounceUntil) {
|
|
229
|
+
if (now < runtime.debounceUntil && !gate.headroomExceeded) {
|
|
224
230
|
runtime.diagCtxDebounce++;
|
|
225
231
|
return tailResult() ?? undefined;
|
|
226
232
|
}
|
|
@@ -109,10 +109,20 @@ function doCompact(
|
|
|
109
109
|
runtime.pulsing = false;
|
|
110
110
|
|
|
111
111
|
if (result.skipped) return { skipped: true };
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
112
|
+
// C1 (v0.21.10): lastCheckpointId tracks "the checkpoint backing this epoch",
|
|
113
|
+
// so it is stamped on BOTH paths — a matched-dedup checkpoint backs this epoch
|
|
114
|
+
// just as much as a freshly created one. Previously the dedup path left it
|
|
115
|
+
// undefined, so a runtime session whose every compaction deduped (common after
|
|
116
|
+
// a process restart, when checkpoints persist but `rt` is rebuilt) never set it
|
|
117
|
+
// → liveTrim's trimCache fell back to result.checkpointId (the matched id) →
|
|
118
|
+
// `trimCache.checkpointId === rt.lastCheckpointId` was `"chkpt_001" !== undefined`
|
|
119
|
+
// → the D.2/D.3 replay NEVER matched and the full pipeline re-ran on every
|
|
120
|
+
// context event (liveTrimReplays: 0, "comp lag warn"). A later fire matching a
|
|
121
|
+
// DIFFERENT checkpoint now changes the key once (one cache regeneration), then
|
|
122
|
+
// replays stabilise. `persistedThisSession` keeps its narrower meaning ("we
|
|
123
|
+
// wrote NEW state this session") and stays gated on !deduped.
|
|
124
|
+
if (!result.deduped) runtime.rt.persistedThisSession = true;
|
|
125
|
+
runtime.rt.lastCheckpointId = result.checkpointId;
|
|
116
126
|
runtime.rt.lastCompactedFrom = result.compactedFrom;
|
|
117
127
|
runtime.rt.lastCompactedTokens = result.tokenEstimate;
|
|
118
128
|
runtime.rt.dedupAttempts++;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-mega-compact",
|
|
3
|
-
"version": "0.21.
|
|
3
|
+
"version": "0.21.11",
|
|
4
4
|
"description": "Layered, local, vector-backed context compressor for pi — supersede/collapse/cluster compaction with deduped inline recall.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "BSD-3-Clause",
|
package/src/config/dedup.ts
CHANGED
|
@@ -66,6 +66,18 @@ export interface DedupConfigShape {
|
|
|
66
66
|
L2_COSINE_CODE: number | null;
|
|
67
67
|
L2_COSINE_PROSE: number | null;
|
|
68
68
|
L1_JACCARD: number; // MinHash/LSH near-dup verification
|
|
69
|
+
/**
|
|
70
|
+
* Degenerate-match guard (incident 2026-08-19). ON declines an L1/L2 collapse
|
|
71
|
+
* when the MATCHED stored checkpoint is a content-free skeleton and the
|
|
72
|
+
* incoming candidate is richer, so a degenerate checkpoint can no longer
|
|
73
|
+
* absorb every future compaction forever. OFF is byte-identical to the
|
|
74
|
+
* pre-guard cascade. See src/dedup/degenerate.ts.
|
|
75
|
+
*/
|
|
76
|
+
DEDUP_DEGENERATE_GUARD: boolean;
|
|
77
|
+
/** Absolute token floor under which a stored summary counts as degenerate. */
|
|
78
|
+
DEDUP_DEGEN_MIN_TOKENS: number;
|
|
79
|
+
/** Relative floor as a fraction of the summary's original region size. */
|
|
80
|
+
DEDUP_DEGEN_MIN_PCT: number;
|
|
69
81
|
DEDUP_SIM: number; // legacy content-similarity fallback
|
|
70
82
|
MMR_LAMBDA: number; // retrieval diversity
|
|
71
83
|
SEMDEDUP_COSINE: number; // offline SemDeDup pair threshold
|
|
@@ -123,6 +135,9 @@ export function loadDedupConfig(): DedupConfigShape {
|
|
|
123
135
|
L2_COSINE_CODE: envNumOrNull("MEGACOMPACT_L2_THRESHOLD_CODE"),
|
|
124
136
|
L2_COSINE_PROSE: envNumOrNull("MEGACOMPACT_L2_THRESHOLD_PROSE"),
|
|
125
137
|
L1_JACCARD: envNum("MEGACOMPACT_L1_JACCARD", 0.8),
|
|
138
|
+
DEDUP_DEGENERATE_GUARD: envBool("MEGACOMPACT_DEDUP_DEGENERATE_GUARD", true),
|
|
139
|
+
DEDUP_DEGEN_MIN_TOKENS: envNum("MEGACOMPACT_DEDUP_DEGEN_MIN_TOKENS", 48),
|
|
140
|
+
DEDUP_DEGEN_MIN_PCT: envNum("MEGACOMPACT_DEDUP_DEGEN_MIN_PCT", 0.005),
|
|
126
141
|
DEDUP_SIM: envNum("MEGACOMPACT_DEDUP_SIM", 0.9),
|
|
127
142
|
MMR_LAMBDA: envNum("MEGACOMPACT_MMR_LAMBDA", 0.5),
|
|
128
143
|
SEMDEDUP_COSINE: envNum("MEGACOMPACT_SEMDEDUP_COSINE", 0.95),
|