@yeaft/webchat-agent 1.0.447 → 1.0.448
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/local-runtime/version.json +1 -1
- package/local-runtime/web/app.bundle.js +1 -1
- package/local-runtime/web/app.bundle.js.gz +0 -0
- package/local-runtime/web/index.html +1 -1
- package/package.json +1 -1
- package/yeaft/archive/turn-archive.js +1 -1
- package/yeaft/cli.js +15 -17
- package/yeaft/config.js +2 -3
- package/yeaft/conversation/internal-control.js +2 -0
- package/yeaft/conversation/persist.js +22 -216
- package/yeaft/conversation/visible-entry.js +1 -1
- package/yeaft/effort.js +0 -2
- package/yeaft/engine.js +31 -376
- package/yeaft/history-window.js +570 -0
- package/yeaft/llm/adapter.js +1 -1
- package/yeaft/llm/anthropic.js +1 -1
- package/yeaft/llm/models-dev.js +2 -1
- package/yeaft/llm/openai-responses.js +1 -1
- package/yeaft/llm/router.js +2 -3
- package/yeaft/llm/usage-accounting.js +2 -2
- package/yeaft/pair-sanitize.js +3 -3
- package/yeaft/prompts.js +3 -4
- package/yeaft/session.js +0 -51
- package/yeaft/stdio-protocol.js +0 -1
- package/yeaft/stop-hooks.js +5 -6
- package/yeaft/turn-utils.js +5 -5
- package/yeaft/web-bridge.js +57 -135
- package/yeaft/compact/compactor.js +0 -283
- package/yeaft/compact/orchestrator.js +0 -141
- package/yeaft/compact/partition.js +0 -94
- package/yeaft/compact/triggers.js +0 -54
- package/yeaft/compact/turn-group.js +0 -85
- package/yeaft/history-compact.js +0 -764
|
@@ -1,283 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* compactor.js — per-group post-turn history compactor.
|
|
3
|
-
*
|
|
4
|
-
* Owns the orchestration that previously lived inline in
|
|
5
|
-
* `agent/yeaft/web-bridge.js` as `getCompactState` /
|
|
6
|
-
* `scheduleCompactAfterTurn` / `runCompactNow`. Bridge keeps WS knowledge
|
|
7
|
-
* and history ownership; this class owns:
|
|
8
|
-
* - per-group single-flight + anti-starvation `pending` chain,
|
|
9
|
-
* - the precheck via `shouldCompactHistory`,
|
|
10
|
-
* - the call into `compactHistory` (which itself calls the supplied
|
|
11
|
-
* `summarize` injectable — bound to `Engine.summarizeForCompact` by
|
|
12
|
-
* `session.js`),
|
|
13
|
-
* - the race-guard that bails when the live history reference / length
|
|
14
|
-
* diverges from the snapshot we captured before awaiting the LLM.
|
|
15
|
-
*
|
|
16
|
-
* History is passed in PER CALL via a `historyHandle = { get, set }` so
|
|
17
|
-
* Compactor never has to know about `sessionContexts`, `historyHydrated`,
|
|
18
|
-
* or any other bridge-internal field. The WS event sink is wired
|
|
19
|
-
* separately via `setOnCompacted`.
|
|
20
|
-
*
|
|
21
|
-
* Engine instances are per-VP-per-group (`vpEngines` keyed by
|
|
22
|
-
* `${sessionId}::${vpId}`), so this orchestration cannot live on `Engine`
|
|
23
|
-
* itself: a per-group single-flight slot pinned to a per-VP Engine is a
|
|
24
|
-
* category error. Compactor is constructed once per session, beside the
|
|
25
|
-
* `dreamScheduler`.
|
|
26
|
-
*/
|
|
27
|
-
|
|
28
|
-
import { compactHistory, shouldCompactHistory } from '../history-compact.js';
|
|
29
|
-
|
|
30
|
-
/**
|
|
31
|
-
* Output cap for the summarizer call. The full compacted history is
|
|
32
|
-
* built around the LLM-produced summary, so the summary itself must be
|
|
33
|
-
* bounded — too long defeats the point of compacting, too short loses
|
|
34
|
-
* context. 1024 tokens is the value that lived inline in the old
|
|
35
|
-
* `runCompactNow` helper in `web-bridge.js` and matches the budget
|
|
36
|
-
* `compactHistory` plans around.
|
|
37
|
-
*/
|
|
38
|
-
const SUMMARIZER_MAX_TOKENS = 1024;
|
|
39
|
-
|
|
40
|
-
/**
|
|
41
|
-
* @typedef {object} CompactorHistoryHandle
|
|
42
|
-
* @property {() => Array<object>} get — return the current live history array
|
|
43
|
-
* @property {(next: Array<object>) => void} set — replace the array reference
|
|
44
|
-
*/
|
|
45
|
-
|
|
46
|
-
/**
|
|
47
|
-
* @typedef {object} CompactedResult
|
|
48
|
-
* @property {string|null} reason
|
|
49
|
-
* @property {number} beforeTurns
|
|
50
|
-
* @property {number} afterTurns
|
|
51
|
-
* @property {number} beforeTokens
|
|
52
|
-
* @property {number} afterTokens
|
|
53
|
-
* @property {number} archivedCount
|
|
54
|
-
*/
|
|
55
|
-
|
|
56
|
-
export class Compactor {
|
|
57
|
-
/**
|
|
58
|
-
* @param {object} opts
|
|
59
|
-
* @param {(args: {system: string, prompt: string, maxTokens?: number}) => Promise<string>} opts.summarize
|
|
60
|
-
* Bound to `engine.summarizeForCompact` by `session.js`. Returns
|
|
61
|
-
* the trimmed summary text (or '' on failure — `compactHistory`
|
|
62
|
-
* treats that as a soft failure).
|
|
63
|
-
* @param {() => number|undefined} [opts.getMaxContextTokens]
|
|
64
|
-
* Returns the model-aware context window (preferred — provided
|
|
65
|
-
* by `session.js` via `resolveContextWindow(model, config)`) or
|
|
66
|
-
* a flat `config.maxContextTokens` fallback. The number is
|
|
67
|
-
* threaded into `shouldCompactHistory` as `maxContextTokens`.
|
|
68
|
-
* @param {() => number|undefined} [opts.getTriggerRatio]
|
|
69
|
-
* Returns the fraction-of-context threshold (e.g. `0.7` for the
|
|
70
|
-
* user-stated "70% of model context"). The application-wide
|
|
71
|
-
* default is 0.7, enforced by the injector in `session.js` — a
|
|
72
|
-
* finite number in (0, 1) wins, anything else falls back to 0.7.
|
|
73
|
-
*
|
|
74
|
-
* When NO injector is wired (typically test fixtures that omit
|
|
75
|
-
* `getTriggerRatio` entirely), this falls through to
|
|
76
|
-
* `shouldCompactHistory`'s library default (`DEFAULT_TOKEN_FRACTION`
|
|
77
|
-
* = 0.5). That gap is intentional: the library default is the
|
|
78
|
-
* documented unit-test contract, the application default is the
|
|
79
|
-
* user-facing product contract, and the injector boundary is what
|
|
80
|
-
* keeps them from drifting in production. Production callers MUST
|
|
81
|
-
* wire `getTriggerRatio` (session.js does). Live-read so a config
|
|
82
|
-
* edit takes effect without reboot.
|
|
83
|
-
* @param {() => string|undefined} [opts.getLanguage]
|
|
84
|
-
* Returns the live `config.language`. Threaded into
|
|
85
|
-
* `compactHistory` so the compactor's summary prompt + the
|
|
86
|
-
* "session continued" wrapper render in the user's preferred
|
|
87
|
-
* locale instead of always English.
|
|
88
|
-
* @param {(sessionId: string, result: CompactedResult) => void} [opts.onCompacted]
|
|
89
|
-
* Optional sink. Bridge wires this to send the
|
|
90
|
-
* `yeaft_history_compacted` WS event. Default: no-op. Can be
|
|
91
|
-
* replaced post-construction via `setOnCompacted`.
|
|
92
|
-
*/
|
|
93
|
-
constructor({ summarize, getMaxContextTokens, getTriggerRatio, getLanguage, onCompacted } = {}) {
|
|
94
|
-
if (typeof summarize !== 'function') {
|
|
95
|
-
throw new TypeError('Compactor: summarize is required');
|
|
96
|
-
}
|
|
97
|
-
this._summarize = summarize;
|
|
98
|
-
this._getMaxContextTokens = typeof getMaxContextTokens === 'function'
|
|
99
|
-
? getMaxContextTokens
|
|
100
|
-
: () => undefined;
|
|
101
|
-
this._getTriggerRatio = typeof getTriggerRatio === 'function'
|
|
102
|
-
? getTriggerRatio
|
|
103
|
-
: () => undefined;
|
|
104
|
-
this._getLanguage = typeof getLanguage === 'function'
|
|
105
|
-
? getLanguage
|
|
106
|
-
: () => undefined;
|
|
107
|
-
this._onCompacted = typeof onCompacted === 'function' ? onCompacted : () => {};
|
|
108
|
-
/** @type {Map<string, { inFlight: Promise<void>|null, pending: boolean }>} */
|
|
109
|
-
this._states = new Map();
|
|
110
|
-
}
|
|
111
|
-
|
|
112
|
-
/**
|
|
113
|
-
* Replace the post-success sink. Bridge calls this from
|
|
114
|
-
* `installYeaftRuntimeBridge` once the bridge-local
|
|
115
|
-
* `sendToServer` + `yeaftConversationId` are available — keeps WS
|
|
116
|
-
* knowledge out of Compactor's constructor and avoids a circular
|
|
117
|
-
* import between `session.js` and `web-bridge.js`.
|
|
118
|
-
*/
|
|
119
|
-
setOnCompacted(fn) {
|
|
120
|
-
this._onCompacted = typeof fn === 'function' ? fn : () => {};
|
|
121
|
-
}
|
|
122
|
-
|
|
123
|
-
/** Get-or-create the per-group state record. */
|
|
124
|
-
_state(sessionId) {
|
|
125
|
-
let s = this._states.get(sessionId);
|
|
126
|
-
if (!s) {
|
|
127
|
-
s = { inFlight: null, pending: false };
|
|
128
|
-
this._states.set(sessionId, s);
|
|
129
|
-
}
|
|
130
|
-
return s;
|
|
131
|
-
}
|
|
132
|
-
|
|
133
|
-
/**
|
|
134
|
-
* Entry-gate await. Bridge calls at the top of `handleYeaftGroupChat`
|
|
135
|
-
* so a brand-new turn never reads a half-mutated history mid-compact.
|
|
136
|
-
* Other groups' compacts never block this gate (per-group keying).
|
|
137
|
-
*
|
|
138
|
-
* @param {string} sessionId
|
|
139
|
-
*/
|
|
140
|
-
async awaitInFlight(sessionId) {
|
|
141
|
-
if (!sessionId) return;
|
|
142
|
-
const s = this._states.get(sessionId);
|
|
143
|
-
if (s && s.inFlight) {
|
|
144
|
-
try { await s.inFlight; } catch { /* _runOnce already logs */ }
|
|
145
|
-
}
|
|
146
|
-
}
|
|
147
|
-
|
|
148
|
-
/**
|
|
149
|
-
* Post-turn fire-and-forget. Bridge calls after the per-VP fanout
|
|
150
|
-
* completes for a turn. The `historyHandle` MUST be a per-call value:
|
|
151
|
-
* its `get` / `set` close over the bridge's sessionId-scoped helpers
|
|
152
|
-
* (`getOrCreateSessionHistory` / `setGroupHistory`), not over a frozen
|
|
153
|
-
* snapshot, so a chained follow-up sees fresh state.
|
|
154
|
-
*
|
|
155
|
-
* Anti-starvation: while a compact is in flight, additional
|
|
156
|
-
* `scheduleAfterTurn` calls collapse to a single `pending=true`. Once
|
|
157
|
-
* the in-flight finishes, exactly one follow-up runs (no matter how
|
|
158
|
-
* many turns piled up during the await).
|
|
159
|
-
*
|
|
160
|
-
* @param {string} sessionId
|
|
161
|
-
* @param {CompactorHistoryHandle} historyHandle
|
|
162
|
-
*/
|
|
163
|
-
scheduleAfterTurn(sessionId, historyHandle) {
|
|
164
|
-
if (!sessionId || !historyHandle) return;
|
|
165
|
-
const s = this._state(sessionId);
|
|
166
|
-
if (s.inFlight) {
|
|
167
|
-
// Anti-starvation: compact is already running. Mark a follow-up
|
|
168
|
-
// so when it finishes, it re-evaluates and runs again if still
|
|
169
|
-
// triggered.
|
|
170
|
-
s.pending = true;
|
|
171
|
-
return;
|
|
172
|
-
}
|
|
173
|
-
s.inFlight = this._runOnce(sessionId, historyHandle).finally(() => {
|
|
174
|
-
s.inFlight = null;
|
|
175
|
-
// If turns piled up while we were running and compaction is still
|
|
176
|
-
// needed, chain a follow-up. Use a microtask so the .finally
|
|
177
|
-
// chain settles cleanly before the next promise is created.
|
|
178
|
-
if (s.pending) {
|
|
179
|
-
s.pending = false;
|
|
180
|
-
queueMicrotask(() => this.scheduleAfterTurn(sessionId, historyHandle));
|
|
181
|
-
}
|
|
182
|
-
});
|
|
183
|
-
}
|
|
184
|
-
|
|
185
|
-
/**
|
|
186
|
-
* One compact pass. Captures snapshot reference + length, runs the
|
|
187
|
-
* precheck, calls `compactHistory`, race-guards, applies via
|
|
188
|
-
* `historyHandle.set`, then notifies `onCompacted`. All errors are
|
|
189
|
-
* swallowed (logged) so `inFlight` always clears.
|
|
190
|
-
*
|
|
191
|
-
* @param {string} sessionId
|
|
192
|
-
* @param {CompactorHistoryHandle} historyHandle
|
|
193
|
-
*/
|
|
194
|
-
async _runOnce(sessionId, historyHandle) {
|
|
195
|
-
try {
|
|
196
|
-
// Capture reference AND length. If a different code path swaps the
|
|
197
|
-
// array (`consolidate` event, session reset, manual clear) the
|
|
198
|
-
// reference will differ. If a driver path push-mutates new
|
|
199
|
-
// messages in place (e.g. `route_forward` triggering a new VP
|
|
200
|
-
// turn during compact), the reference is the same but the length
|
|
201
|
-
// grew. Both cases mean our snapshot is no longer canonical.
|
|
202
|
-
const snapshot = historyHandle.get();
|
|
203
|
-
if (!Array.isArray(snapshot) || snapshot.length === 0) return;
|
|
204
|
-
const snapshotLen = snapshot.length;
|
|
205
|
-
|
|
206
|
-
const maxContextTokens = this._getMaxContextTokens();
|
|
207
|
-
const tokenFraction = this._getTriggerRatio();
|
|
208
|
-
|
|
209
|
-
// Cheap O(n) precheck so we don't bother engaging the LLM at all
|
|
210
|
-
// when the conversation is still small. `compactHistory` runs the
|
|
211
|
-
// same check internally, but only after building the summarizer
|
|
212
|
-
// input — this keeps small chats off the LLM altogether.
|
|
213
|
-
//
|
|
214
|
-
// `tokenFraction` is the user-configurable ratio knob — production
|
|
215
|
-
// wiring (session.js) ALWAYS returns a finite (0,1) number defaulting
|
|
216
|
-
// to 0.7. The `undefined` branch (helper falls back to its
|
|
217
|
-
// `DEFAULT_TOKEN_FRACTION = 0.5`) is reachable only when a caller
|
|
218
|
-
// skips `getTriggerRatio` entirely — see constructor JSDoc.
|
|
219
|
-
const triage = shouldCompactHistory(snapshot, { maxContextTokens, tokenFraction });
|
|
220
|
-
if (!triage.trigger) return;
|
|
221
|
-
|
|
222
|
-
const summarize = ({ system, prompt }) =>
|
|
223
|
-
this._summarize({ system, prompt, maxTokens: SUMMARIZER_MAX_TOKENS });
|
|
224
|
-
|
|
225
|
-
const result = await compactHistory(snapshot, {
|
|
226
|
-
summarize,
|
|
227
|
-
maxContextTokens,
|
|
228
|
-
tokenFraction,
|
|
229
|
-
language: this._getLanguage(),
|
|
230
|
-
});
|
|
231
|
-
if (!result || !result.compacted) {
|
|
232
|
-
if (result && result.error) {
|
|
233
|
-
console.warn(
|
|
234
|
-
`[Yeaft] history compact: summarizer failed (${result.error}); ` +
|
|
235
|
-
`keeping ${result.beforeTurns} turns / ~${result.beforeTokens} tokens`
|
|
236
|
-
);
|
|
237
|
-
}
|
|
238
|
-
return;
|
|
239
|
-
}
|
|
240
|
-
|
|
241
|
-
// Race guard: if the group's history was reassigned during the
|
|
242
|
-
// await (e.g. consolidate / reset), or push-mutated by a driver
|
|
243
|
-
// path, do NOT overwrite the fresh state with our stale compacted
|
|
244
|
-
// snapshot. These discards are EXPECTED during normal session
|
|
245
|
-
// resets / route_forward bursts — log at debug, not info.
|
|
246
|
-
const current = historyHandle.get();
|
|
247
|
-
if (current !== snapshot) {
|
|
248
|
-
console.debug('[Yeaft] history compact: history was reset during compact — discarding stale summary');
|
|
249
|
-
return;
|
|
250
|
-
}
|
|
251
|
-
if (current.length !== snapshotLen) {
|
|
252
|
-
console.debug('[Yeaft] history compact: history was appended-to during compact — discarding stale summary');
|
|
253
|
-
return;
|
|
254
|
-
}
|
|
255
|
-
|
|
256
|
-
historyHandle.set(result.messages);
|
|
257
|
-
console.log(
|
|
258
|
-
`[Yeaft] history compacted (reason=${result.reason}): ` +
|
|
259
|
-
`turns ${result.beforeTurns}→${result.afterTurns}, ` +
|
|
260
|
-
`tokens ~${result.beforeTokens}→${result.afterTokens}, ` +
|
|
261
|
-
`archived ${result.archivedCount} messages`
|
|
262
|
-
);
|
|
263
|
-
|
|
264
|
-
try {
|
|
265
|
-
this._onCompacted(sessionId, {
|
|
266
|
-
reason: result.reason,
|
|
267
|
-
beforeTurns: result.beforeTurns,
|
|
268
|
-
afterTurns: result.afterTurns,
|
|
269
|
-
beforeTokens: result.beforeTokens,
|
|
270
|
-
afterTokens: result.afterTokens,
|
|
271
|
-
archivedCount: result.archivedCount,
|
|
272
|
-
});
|
|
273
|
-
} catch { /* sink failure must not abort the orchestrator */ }
|
|
274
|
-
} catch (err) {
|
|
275
|
-
console.warn(`[Yeaft] history compact: unexpected failure (sessionId=${sessionId})`, err?.message || err);
|
|
276
|
-
}
|
|
277
|
-
}
|
|
278
|
-
|
|
279
|
-
/** Test-only: clear all per-group state. */
|
|
280
|
-
__testReset() {
|
|
281
|
-
this._states.clear();
|
|
282
|
-
}
|
|
283
|
-
}
|
|
@@ -1,141 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* compact/orchestrator.js — DESIGN.md §4.2.
|
|
3
|
-
*
|
|
4
|
-
* One trigger, one pass, three tracks. The orchestrator owns the
|
|
5
|
-
* sequencing; the actual LLM-driven summarisation and extraction are
|
|
6
|
-
* supplied as injectables so this file stays small, deterministic, and
|
|
7
|
-
* testable without network.
|
|
8
|
-
*
|
|
9
|
-
* Track 1 — message compaction (always runs):
|
|
10
|
-
* 1. Find the cooling turn-groups (older than the hot window).
|
|
11
|
-
* 2. Generate a `compact_summary` of those groups.
|
|
12
|
-
* 3. Archive each cooling group atomically; replace it in the live
|
|
13
|
-
* messages array with a single placeholder message.
|
|
14
|
-
*
|
|
15
|
-
* Track 2 — task summary refresh (when `taskId` is provided):
|
|
16
|
-
* 4. Refresh `tasks/<tid>/summary.md` from the cooling groups + the
|
|
17
|
-
* prior summary. Atomic via the existing `writeSummary` helper.
|
|
18
|
-
*
|
|
19
|
-
* Track 3 — memory extraction (always runs):
|
|
20
|
-
* 5. Extract durable facts/lessons/preferences and write them as
|
|
21
|
-
* scope entries through the supplied `extract` callback. The
|
|
22
|
-
* callback owns scope routing and `index.md` upserts; this file
|
|
23
|
-
* just hands it the cooling groups.
|
|
24
|
-
*
|
|
25
|
-
* Atomicity rule (§9.2): each cooling turn-group is archived as a
|
|
26
|
-
* unit. We never break a `[user, assistant(toolCalls), tool…]` triple.
|
|
27
|
-
* Track 1 is the only place that mutates `messages`.
|
|
28
|
-
*/
|
|
29
|
-
|
|
30
|
-
import { groupTurns, pickCoolingGroups, indicesFromGroups } from './turn-group.js';
|
|
31
|
-
|
|
32
|
-
/**
|
|
33
|
-
* @typedef {{
|
|
34
|
-
* summarise: (coolingMessages: object[]) => Promise<string>,
|
|
35
|
-
* archive: (groupIndex: number, coolingMessages: object[]) => Promise<{ turnId: string }>,
|
|
36
|
-
* extract?: (coolingMessages: object[]) => Promise<{ written: number }>,
|
|
37
|
-
* refreshTaskSummary?: (coolingMessages: object[], priorSummary: string) => Promise<string>,
|
|
38
|
-
* readPriorTaskSummary?: () => Promise<string>,
|
|
39
|
-
* }} CompactHooks
|
|
40
|
-
*/
|
|
41
|
-
|
|
42
|
-
/**
|
|
43
|
-
* @param {{
|
|
44
|
-
* messages: object[],
|
|
45
|
-
* keepHot?: number,
|
|
46
|
-
* taskId?: string | null,
|
|
47
|
-
* root?: string,
|
|
48
|
-
* hooks: CompactHooks,
|
|
49
|
-
* }} args
|
|
50
|
-
* @returns {Promise<{
|
|
51
|
-
* archivedGroups: number,
|
|
52
|
-
* archivedMessages: number,
|
|
53
|
-
* compactSummary: string,
|
|
54
|
-
* extractedCount: number,
|
|
55
|
-
* taskSummaryRefreshed: boolean,
|
|
56
|
-
* nextMessages: object[],
|
|
57
|
-
* }>}
|
|
58
|
-
*/
|
|
59
|
-
export async function runCompact({ messages, keepHot = 10, hooks }) {
|
|
60
|
-
if (!Array.isArray(messages)) {
|
|
61
|
-
throw new Error('runCompact: messages array required');
|
|
62
|
-
}
|
|
63
|
-
if (!hooks || typeof hooks !== 'object') {
|
|
64
|
-
throw new Error('runCompact: hooks required');
|
|
65
|
-
}
|
|
66
|
-
if (typeof hooks.summarise !== 'function') {
|
|
67
|
-
throw new Error('runCompact: hooks.summarise required');
|
|
68
|
-
}
|
|
69
|
-
if (typeof hooks.archive !== 'function') {
|
|
70
|
-
throw new Error('runCompact: hooks.archive required');
|
|
71
|
-
}
|
|
72
|
-
|
|
73
|
-
const groups = groupTurns(messages);
|
|
74
|
-
const { hot, cooling } = pickCoolingGroups(groups, keepHot);
|
|
75
|
-
|
|
76
|
-
// Nothing to compact — return early. We still report `nextMessages` so
|
|
77
|
-
// callers can treat the result uniformly (no copy unless we changed
|
|
78
|
-
// anything).
|
|
79
|
-
if (cooling.length === 0) {
|
|
80
|
-
return {
|
|
81
|
-
archivedGroups: 0,
|
|
82
|
-
archivedMessages: 0,
|
|
83
|
-
compactSummary: '',
|
|
84
|
-
extractedCount: 0,
|
|
85
|
-
taskSummaryRefreshed: false,
|
|
86
|
-
nextMessages: messages,
|
|
87
|
-
};
|
|
88
|
-
}
|
|
89
|
-
|
|
90
|
-
// Slice out the cooling messages once for the summariser/extractor.
|
|
91
|
-
const coolingIdx = indicesFromGroups(cooling);
|
|
92
|
-
const coolingMessages = coolingIdx.map(i => messages[i]);
|
|
93
|
-
|
|
94
|
-
// Track 1.2 — summarise.
|
|
95
|
-
const compactSummary = await hooks.summarise(coolingMessages);
|
|
96
|
-
|
|
97
|
-
// Track 1.3 — archive each cooling group atomically.
|
|
98
|
-
const archiveResults = [];
|
|
99
|
-
for (let i = 0; i < cooling.length; i += 1) {
|
|
100
|
-
const g = cooling[i];
|
|
101
|
-
const groupMsgs = messages.slice(g.start, g.end);
|
|
102
|
-
const r = await hooks.archive(i, groupMsgs);
|
|
103
|
-
archiveResults.push({ ...g, turnId: r?.turnId });
|
|
104
|
-
}
|
|
105
|
-
|
|
106
|
-
// Track 2 — task summary refresh: removed. The legacy `feature/<id>` root
|
|
107
|
-
// scope was dropped along with the Feature system (2026-05-13); under the
|
|
108
|
-
// group-isolated layout feature summaries would live at
|
|
109
|
-
// `group/<g>/feature/<id>/` and are written by dream, not by post-turn
|
|
110
|
-
// compact. Engine no longer passes `taskId`/`root` to this orchestrator.
|
|
111
|
-
const taskSummaryRefreshed = false;
|
|
112
|
-
|
|
113
|
-
// Track 3 — memory extraction.
|
|
114
|
-
let extractedCount = 0;
|
|
115
|
-
if (typeof hooks.extract === 'function') {
|
|
116
|
-
const extractResult = await hooks.extract(coolingMessages);
|
|
117
|
-
if (extractResult && Number.isFinite(extractResult.written)) {
|
|
118
|
-
extractedCount = extractResult.written;
|
|
119
|
-
}
|
|
120
|
-
}
|
|
121
|
-
|
|
122
|
-
// Track 1.3 (cont.) — produce the new messages array with the cooling
|
|
123
|
-
// window replaced by a single `compact_summary` placeholder.
|
|
124
|
-
const placeholder = {
|
|
125
|
-
role: 'system',
|
|
126
|
-
kind: 'compact_summary',
|
|
127
|
-
content: compactSummary || '',
|
|
128
|
-
};
|
|
129
|
-
// Hot starts at the index right after the last cooling group.
|
|
130
|
-
const cutoff = cooling[cooling.length - 1].end;
|
|
131
|
-
const nextMessages = [placeholder, ...messages.slice(cutoff)];
|
|
132
|
-
|
|
133
|
-
return {
|
|
134
|
-
archivedGroups: cooling.length,
|
|
135
|
-
archivedMessages: coolingIdx.length,
|
|
136
|
-
compactSummary: compactSummary || '',
|
|
137
|
-
extractedCount,
|
|
138
|
-
taskSummaryRefreshed,
|
|
139
|
-
nextMessages,
|
|
140
|
-
};
|
|
141
|
-
}
|
|
@@ -1,94 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* compact/partition.js — Hot-window budget partitioning utilities.
|
|
3
|
-
*
|
|
4
|
-
* (Renamed from `agent/yeaft/memory/consolidate.js` on 2026-06-09.) The
|
|
5
|
-
* legacy "consolidate" name and the `memory/` location both pointed at
|
|
6
|
-
* a single concept — Layer-A memory consolidation — that has since been
|
|
7
|
-
* cleanly split:
|
|
8
|
-
*
|
|
9
|
-
* - Memory consolidation / system-prompt maintenance is owned by
|
|
10
|
-
* Dream V2 (per-group diff -> triage -> merge by target scope ->
|
|
11
|
-
* apply via segment-store + summary-store). NONE of that lives here.
|
|
12
|
-
*
|
|
13
|
-
* - Conversation history compaction (the thing this file ACTUALLY
|
|
14
|
-
* serves) is owned by `compact/orchestrator.js`. The two functions
|
|
15
|
-
* below — `shouldCompact` (the "hot window over budget?" predicate
|
|
16
|
-
* that gates `compact/orchestrator.js`) and `partitionMessages`
|
|
17
|
-
* (hot/cold split by token budget) are pure helpers for that
|
|
18
|
-
* orchestrator.
|
|
19
|
-
*
|
|
20
|
-
* Why the move matters: keeping these under `memory/` invited the next
|
|
21
|
-
* person to think "this is part of the memory subsystem" and reach for
|
|
22
|
-
* it during a Dream-v2 patch — the exact category error
|
|
23
|
-
* `DESIGN-COMPACT-VS-DREAM.md` (sibling doc) warns against. Putting
|
|
24
|
-
* them next to `compact/orchestrator.js` makes the ownership obvious
|
|
25
|
-
* from the file tree.
|
|
26
|
-
*/
|
|
27
|
-
|
|
28
|
-
// ─── Constants ──────────────────────────────────────────────────
|
|
29
|
-
|
|
30
|
-
/** Default MESSAGE_TOKEN_BUDGET for hot message compaction. */
|
|
31
|
-
export const DEFAULT_MESSAGE_TOKEN_BUDGET = 32768;
|
|
32
|
-
|
|
33
|
-
/** After compact, keep this fraction of the budget. */
|
|
34
|
-
export const COMPACT_KEEP_RATIO = 0.4;
|
|
35
|
-
|
|
36
|
-
/** Minimum messages to keep hot (newest). */
|
|
37
|
-
const MIN_KEEP_MESSAGES = 3;
|
|
38
|
-
|
|
39
|
-
/**
|
|
40
|
-
* Check if a compact pass should be triggered.
|
|
41
|
-
*
|
|
42
|
-
* Semantically: "is the hot window over budget?" — the predicate that
|
|
43
|
-
* gates `compact/orchestrator.js`. Renamed from `shouldConsolidate` on
|
|
44
|
-
* 2026-06-09; the old name leaked Dream V2's vocabulary into a file
|
|
45
|
-
* that exclusively serves Compact.
|
|
46
|
-
*
|
|
47
|
-
* @param {import('../conversation/persist.js').ConversationStore} conversationStore
|
|
48
|
-
* @param {number} [budget] — MESSAGE_TOKEN_BUDGET
|
|
49
|
-
* @returns {boolean}
|
|
50
|
-
*/
|
|
51
|
-
export function shouldCompact(conversationStore, budget = DEFAULT_MESSAGE_TOKEN_BUDGET) {
|
|
52
|
-
const hotTokens = conversationStore.hotTokens();
|
|
53
|
-
return hotTokens > budget;
|
|
54
|
-
}
|
|
55
|
-
|
|
56
|
-
/**
|
|
57
|
-
* Determine which messages to archive (move to cold).
|
|
58
|
-
* Strategy: from oldest, accumulate tokens until remaining ≤ budget * 40%.
|
|
59
|
-
* Always keep at least MIN_KEEP_MESSAGES.
|
|
60
|
-
*
|
|
61
|
-
* @param {object[]} messages — all hot messages, sorted chronologically
|
|
62
|
-
* @param {number} budget — MESSAGE_TOKEN_BUDGET
|
|
63
|
-
* @returns {{ toArchive: object[], toKeep: object[] }}
|
|
64
|
-
*/
|
|
65
|
-
export function partitionMessages(messages, budget = DEFAULT_MESSAGE_TOKEN_BUDGET) {
|
|
66
|
-
if (messages.length <= MIN_KEEP_MESSAGES) {
|
|
67
|
-
return { toArchive: [], toKeep: messages };
|
|
68
|
-
}
|
|
69
|
-
|
|
70
|
-
const keepBudget = Math.floor(budget * COMPACT_KEEP_RATIO);
|
|
71
|
-
|
|
72
|
-
// Work backwards from newest: accumulate tokens until we hit keepBudget
|
|
73
|
-
let keepTokens = 0;
|
|
74
|
-
let keepStart = messages.length;
|
|
75
|
-
|
|
76
|
-
for (let i = messages.length - 1; i >= 0; i--) {
|
|
77
|
-
const msgTokens = messages[i].tokens_est || 0;
|
|
78
|
-
if (keepTokens + msgTokens > keepBudget && (messages.length - i) >= MIN_KEEP_MESSAGES) {
|
|
79
|
-
keepStart = i + 1;
|
|
80
|
-
break;
|
|
81
|
-
}
|
|
82
|
-
keepTokens += msgTokens;
|
|
83
|
-
if (i === 0) keepStart = 0;
|
|
84
|
-
}
|
|
85
|
-
|
|
86
|
-
// Ensure at least MIN_KEEP_MESSAGES are kept
|
|
87
|
-
keepStart = Math.min(keepStart, messages.length - MIN_KEEP_MESSAGES);
|
|
88
|
-
keepStart = Math.max(keepStart, 0);
|
|
89
|
-
|
|
90
|
-
return {
|
|
91
|
-
toArchive: messages.slice(0, keepStart),
|
|
92
|
-
toKeep: messages.slice(keepStart),
|
|
93
|
-
};
|
|
94
|
-
}
|
|
@@ -1,54 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* compact/triggers.js — DESIGN.md §4.1.
|
|
3
|
-
*
|
|
4
|
-
* Pure functions: do I need to compact, and which triggers fired?
|
|
5
|
-
* Caller (orchestrator) decides what to do next; this module never
|
|
6
|
-
* touches disk or messages.
|
|
7
|
-
*/
|
|
8
|
-
|
|
9
|
-
const DEFAULT_MAX_MESSAGES = 50;
|
|
10
|
-
const DEFAULT_TOKEN_RATIO = 0.9;
|
|
11
|
-
const DEFAULT_IDLE_MS = 2 * 60 * 1000;
|
|
12
|
-
|
|
13
|
-
/**
|
|
14
|
-
* @param {{
|
|
15
|
-
* messages: object[],
|
|
16
|
-
* tokenCount: number,
|
|
17
|
-
* contextLimit: number,
|
|
18
|
-
* lastActivityAt?: number,
|
|
19
|
-
* now?: number,
|
|
20
|
-
* explicit?: boolean,
|
|
21
|
-
* maxMessages?: number,
|
|
22
|
-
* tokenRatio?: number,
|
|
23
|
-
* idleMs?: number,
|
|
24
|
-
* }} state
|
|
25
|
-
* @returns {{ trigger: boolean, reasons: string[] }}
|
|
26
|
-
*/
|
|
27
|
-
export function evaluateCompactTriggers(state = {}) {
|
|
28
|
-
const reasons = [];
|
|
29
|
-
const messages = Array.isArray(state.messages) ? state.messages : [];
|
|
30
|
-
const tokenCount = Number.isFinite(state.tokenCount) ? state.tokenCount : 0;
|
|
31
|
-
const contextLimit = Number.isFinite(state.contextLimit) && state.contextLimit > 0
|
|
32
|
-
? state.contextLimit : 0;
|
|
33
|
-
const tokenRatio = Number.isFinite(state.tokenRatio) ? state.tokenRatio : DEFAULT_TOKEN_RATIO;
|
|
34
|
-
const maxMessages = Number.isFinite(state.maxMessages) ? state.maxMessages : DEFAULT_MAX_MESSAGES;
|
|
35
|
-
const idleMs = Number.isFinite(state.idleMs) ? state.idleMs : DEFAULT_IDLE_MS;
|
|
36
|
-
|
|
37
|
-
if (state.explicit) reasons.push('explicit');
|
|
38
|
-
|
|
39
|
-
if (contextLimit > 0 && tokenCount > tokenRatio * contextLimit) {
|
|
40
|
-
reasons.push('token_threshold');
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
if (messages.length > maxMessages) {
|
|
44
|
-
reasons.push('message_count');
|
|
45
|
-
}
|
|
46
|
-
|
|
47
|
-
if (Number.isFinite(state.lastActivityAt) && Number.isFinite(state.now)) {
|
|
48
|
-
if (state.now - state.lastActivityAt > idleMs) {
|
|
49
|
-
reasons.push('idle');
|
|
50
|
-
}
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
return { trigger: reasons.length > 0, reasons };
|
|
54
|
-
}
|
|
@@ -1,85 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* compact/turn-group.js — DESIGN.md §9.2.
|
|
3
|
-
*
|
|
4
|
-
* Group messages into turns whose unit of archiving is "atomic": each
|
|
5
|
-
* group is either kept entirely live or archived entirely. Archiving
|
|
6
|
-
* an assistant message that contained `toolCalls` while leaving the
|
|
7
|
-
* tool results live (or vice versa) breaks the OpenAI invariant that
|
|
8
|
-
* `tool_call_id`s must be paired.
|
|
9
|
-
*
|
|
10
|
-
* A "turn group" starts on a `user` message and extends through every
|
|
11
|
-
* subsequent assistant + tool message until the next user message. The
|
|
12
|
-
* grouping function returns an array of `{ start, end, indices }` pairs
|
|
13
|
-
* where `[start, end)` is a half-open range over the input array.
|
|
14
|
-
*
|
|
15
|
-
* Edge cases:
|
|
16
|
-
* - Leading non-user messages (e.g. a system or tool prelude written
|
|
17
|
-
* by an init hook) form a group of their own at index 0.
|
|
18
|
-
* - Trailing assistant/tool messages (incomplete turn) form the final
|
|
19
|
-
* group — same rule.
|
|
20
|
-
* - Empty input → empty result.
|
|
21
|
-
*/
|
|
22
|
-
|
|
23
|
-
/**
|
|
24
|
-
* @param {object[]} messages
|
|
25
|
-
* @returns {Array<{ start: number, end: number, role: string }>}
|
|
26
|
-
* `role` reflects the group's anchor (the `user` message, if any;
|
|
27
|
-
* otherwise the first message in the group).
|
|
28
|
-
*/
|
|
29
|
-
export function groupTurns(messages) {
|
|
30
|
-
if (!Array.isArray(messages) || messages.length === 0) return [];
|
|
31
|
-
const groups = [];
|
|
32
|
-
let cur = { start: 0, end: 0, role: messages[0]?.role || 'unknown' };
|
|
33
|
-
for (let i = 0; i < messages.length; i += 1) {
|
|
34
|
-
const m = messages[i];
|
|
35
|
-
if (!m || typeof m !== 'object') {
|
|
36
|
-
// Treat as belonging to the current group (don't break pairing).
|
|
37
|
-
cur.end = i + 1;
|
|
38
|
-
continue;
|
|
39
|
-
}
|
|
40
|
-
if (m.role === 'user' && cur.end > cur.start) {
|
|
41
|
-
groups.push(cur);
|
|
42
|
-
cur = { start: i, end: i + 1, role: 'user' };
|
|
43
|
-
} else {
|
|
44
|
-
cur.end = i + 1;
|
|
45
|
-
if (m.role === 'user') cur.role = 'user';
|
|
46
|
-
}
|
|
47
|
-
}
|
|
48
|
-
if (cur.end > cur.start) groups.push(cur);
|
|
49
|
-
return groups;
|
|
50
|
-
}
|
|
51
|
-
|
|
52
|
-
/**
|
|
53
|
-
* Pick the cut point: keep `keepHot` newest groups live, return the
|
|
54
|
-
* rest as "cooling" candidates for archive.
|
|
55
|
-
*
|
|
56
|
-
* Returns: `{ hot: groups[], cooling: groups[] }`. Both arrays use the
|
|
57
|
-
* same `{start, end, role}` shape from `groupTurns`.
|
|
58
|
-
*
|
|
59
|
-
* @param {Array<{start: number, end: number, role: string}>} groups
|
|
60
|
-
* @param {number} keepHot
|
|
61
|
-
*/
|
|
62
|
-
export function pickCoolingGroups(groups, keepHot = 10) {
|
|
63
|
-
if (!Array.isArray(groups)) return { hot: [], cooling: [] };
|
|
64
|
-
const k = Math.max(0, Math.floor(keepHot));
|
|
65
|
-
if (groups.length <= k) return { hot: groups.slice(), cooling: [] };
|
|
66
|
-
const cut = groups.length - k;
|
|
67
|
-
return {
|
|
68
|
-
hot: groups.slice(cut),
|
|
69
|
-
cooling: groups.slice(0, cut),
|
|
70
|
-
};
|
|
71
|
-
}
|
|
72
|
-
|
|
73
|
-
/**
|
|
74
|
-
* Flatten a list of groups back into the underlying message indices.
|
|
75
|
-
*
|
|
76
|
-
* @param {Array<{start: number, end: number}>} groups
|
|
77
|
-
* @returns {number[]}
|
|
78
|
-
*/
|
|
79
|
-
export function indicesFromGroups(groups) {
|
|
80
|
-
const out = [];
|
|
81
|
-
for (const g of groups) {
|
|
82
|
-
for (let i = g.start; i < g.end; i += 1) out.push(i);
|
|
83
|
-
}
|
|
84
|
-
return out;
|
|
85
|
-
}
|