@cohortapp/agent-sdk 2.16.0 → 2.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +5 -2
- package/docs/guides/poller-daemon-setup.md +4 -1
- package/lib/context/budget.mjs +327 -0
- package/lib/context/budget.test.mjs +252 -0
- package/lib/context/history-scope.mjs +138 -0
- package/lib/context/history-scope.test.mjs +79 -0
- package/lib/model-router/economics.mjs +9 -0
- package/lib/model-router/resolve.mjs +6 -0
- package/lib/org/inbound/facts.mjs +4 -2
- package/lib/org/inbound/hydrate.mjs +555 -51
- package/lib/org/inbound/hydrate.test.mjs +456 -1
- package/package.json +3 -1
- package/scripts/daemon/context-compiler.mjs +52 -21
- package/scripts/daemon/context-compiler.test.mjs +106 -0
- package/scripts/daemon/dispatcher-session-continuity.test.mjs +365 -0
- package/scripts/daemon/dispatcher.mjs +210 -9
- package/scripts/daemon/lib/session-router.mjs +310 -42
- package/scripts/daemon/lib/session-router.test.mjs +260 -1
- package/scripts/daemon/prompt-builder.mjs +97 -12
- package/scripts/daemon/prompt-builder.test.mjs +219 -7
- package/scripts/daemon/responder-history.test.mjs +37 -1
- package/scripts/daemon/responder.mjs +71 -69
package/.env.example
CHANGED
|
@@ -226,6 +226,9 @@ GREPTILE_API_KEY=
|
|
|
226
226
|
# Flags that control system behaviour. These are not secrets.
|
|
227
227
|
#
|
|
228
228
|
|
|
229
|
-
#
|
|
230
|
-
#
|
|
229
|
+
# The session context compiler: pre-compile context before each Claude Code
|
|
230
|
+
# session. ON BY DEFAULT in code as well as here — the two used to disagree
|
|
231
|
+
# (code read `=== "1"`, this file shipped `=1`), so which of two materially
|
|
232
|
+
# different context paths ran depended on whether the operator had copied this
|
|
233
|
+
# file. Set to 0 only to fall back to the legacy path in a hurry.
|
|
231
234
|
DAEMON_CONTEXT_COMPILER=1
|
|
@@ -279,7 +279,10 @@ Pre-compiles session context to reduce prompt size:
|
|
|
279
279
|
|
|
280
280
|
- Reads recent interactions, queue state, and active items
|
|
281
281
|
- Compresses context to fit within token limits
|
|
282
|
-
-
|
|
282
|
+
- **On by default.** Set `DAEMON_CONTEXT_COMPILER=0` in `.env` to fall back to
|
|
283
|
+
the legacy per-prompt history path. The code used to default this OFF while
|
|
284
|
+
`.env.example` shipped `=1`, so which path ran depended on whether the
|
|
285
|
+
operator had copied the example env; both now agree.
|
|
283
286
|
|
|
284
287
|
---
|
|
285
288
|
|
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lib/context/budget.mjs — the declared token budget for an inbound prompt.
|
|
3
|
+
*
|
|
4
|
+
* Every bound in this repo up to now has been a row count or a `.slice(n)` char
|
|
5
|
+
* clamp (design §5.4, R8). Both are proxies: a row count cannot tell a one-word
|
|
6
|
+
* turn from a pasted stack trace, and a char clamp spends the allowance on
|
|
7
|
+
* whichever section happens to be assembled first. So the prompt that reaches
|
|
8
|
+
* the model is whatever the assembly order happened to produce, and the parts
|
|
9
|
+
* that fall off do so SILENTLY — the model reads a thread with a hole in it and
|
|
10
|
+
* cannot tell the hole from a conversation that never happened.
|
|
11
|
+
*
|
|
12
|
+
* This module states the budget instead of discovering it:
|
|
13
|
+
*
|
|
14
|
+
* 1. a budget PER TIER (§5.1) — an Answer gets 6k input tokens, Work 12k,
|
|
15
|
+
* Plan 20k. Context rot is measured: accuracy falls with input length well
|
|
16
|
+
* before the window limit, so a bigger prompt is not a better one.
|
|
17
|
+
* 2. sections filled in PRIORITY order ({@link SECTION_ORDER}) — the trigger
|
|
18
|
+
* is the thing being answered and is never dropped whole; the backlog is
|
|
19
|
+
* the first thing to go.
|
|
20
|
+
* 3. every drop VISIBLE — `… 14 earlier turns not shown`, in the prompt, where
|
|
21
|
+
* the model reads it. A section that did not fit at all still leaves a line
|
|
22
|
+
* saying so.
|
|
23
|
+
*
|
|
24
|
+
* PURE. No clock, no env, no fs, no network: tokens are estimated from
|
|
25
|
+
* character length at {@link CHARS_PER_TOKEN}, the same ratio
|
|
26
|
+
* `context-compiler.mjs` already uses, so the two agree. A caller that has a
|
|
27
|
+
* real tokeniser passes `estimate` and this module uses it instead.
|
|
28
|
+
*
|
|
29
|
+
* Char clamps stay where they are as a second belt — this is the first one.
|
|
30
|
+
*
|
|
31
|
+
* @module lib/context/budget
|
|
32
|
+
*/
|
|
33
|
+
|
|
34
|
+
"use strict";
|
|
35
|
+
|
|
36
|
+
/** Characters per token. The estimate `scripts/daemon/context-compiler.mjs` uses. */
|
|
37
|
+
export const CHARS_PER_TOKEN = 4;
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Input-token budget per reply tier (design §5.4). A tier is the ack/effort
|
|
41
|
+
* tier from `lib/assurance/tier.mjs` — answer | work | plan.
|
|
42
|
+
*/
|
|
43
|
+
export const TIER_BUDGETS = Object.freeze({
|
|
44
|
+
answer: 6000,
|
|
45
|
+
work: 12000,
|
|
46
|
+
plan: 20000,
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* FILL priority. Earlier wins the budget; later is dropped first.
|
|
51
|
+
*
|
|
52
|
+
* trigger the message being answered — without it there is nothing to answer
|
|
53
|
+
* thread the conversation it sits in
|
|
54
|
+
* entity the anchored artifact: the task card, the doc body, the decision
|
|
55
|
+
* digest the distilled record of this thread
|
|
56
|
+
* memory recalled facts from elsewhere
|
|
57
|
+
* backlog what else is open — useful, never load-bearing
|
|
58
|
+
*/
|
|
59
|
+
export const FILL_PRIORITY = Object.freeze([
|
|
60
|
+
"trigger",
|
|
61
|
+
"thread",
|
|
62
|
+
"entity",
|
|
63
|
+
"digest",
|
|
64
|
+
"memory",
|
|
65
|
+
"backlog",
|
|
66
|
+
]);
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* RENDER order — a DIFFERENT question, and conflating the two costs the cache.
|
|
70
|
+
*
|
|
71
|
+
* Fill priority asks "who wins the budget"; the trigger wins it, because a
|
|
72
|
+
* prompt without the message being answered is not a smaller prompt but a
|
|
73
|
+
* broken one. Render order asks "what byte comes first", and design §4 is
|
|
74
|
+
* explicit about the consequence: "any byte change before a breakpoint
|
|
75
|
+
* invalidates everything after it". Rendering the trigger first — the ONE
|
|
76
|
+
* section that is different on every single inbound — gives the block a
|
|
77
|
+
* zero-length stable prefix, so WP-4's `cache_control` breakpoint would sit
|
|
78
|
+
* after volatile content and every read would be a silent miss. Prompt caching
|
|
79
|
+
* is the highest-ROI latency lever in the research pass; this is what it needs.
|
|
80
|
+
*
|
|
81
|
+
* So: stable first, volatile last, trigger last of all. `thread` sits second to
|
|
82
|
+
* last because it is append-only and rendered oldest-first, which makes its own
|
|
83
|
+
* prefix stable between turns even as it grows.
|
|
84
|
+
*/
|
|
85
|
+
export const RENDER_ORDER = Object.freeze([
|
|
86
|
+
"entity",
|
|
87
|
+
"digest",
|
|
88
|
+
"backlog",
|
|
89
|
+
"memory",
|
|
90
|
+
"thread",
|
|
91
|
+
"trigger",
|
|
92
|
+
]);
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* @deprecated Read {@link FILL_PRIORITY} or {@link RENDER_ORDER} by name — the
|
|
96
|
+
* whole point of the split is that "the section order" is two different orders.
|
|
97
|
+
* Kept as an alias so an existing importer does not silently get the wrong one.
|
|
98
|
+
*/
|
|
99
|
+
export const SECTION_ORDER = FILL_PRIORITY;
|
|
100
|
+
|
|
101
|
+
/** Sections that must render SOMETHING rather than be dropped whole. */
|
|
102
|
+
const NEVER_DROPPED_WHOLE = new Set(["trigger"]);
|
|
103
|
+
|
|
104
|
+
/** A blank-safe string. */
|
|
105
|
+
function s(v) {
|
|
106
|
+
return v == null ? "" : String(v);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** The declared budget for a tier; an unknown tier takes the middle one. */
|
|
110
|
+
export function budgetFor(tier) {
|
|
111
|
+
const key = s(tier).toLowerCase();
|
|
112
|
+
return Object.prototype.hasOwnProperty.call(TIER_BUDGETS, key) ? TIER_BUDGETS[key] : TIER_BUDGETS.work;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** Estimated tokens for a string, rounded up so a partial token still costs one. */
|
|
116
|
+
export function estimateTokens(text, charsPerToken = CHARS_PER_TOKEN) {
|
|
117
|
+
const t = s(text);
|
|
118
|
+
if (!t) return 0;
|
|
119
|
+
const per = Number.isFinite(charsPerToken) && charsPerToken > 0 ? charsPerToken : CHARS_PER_TOKEN;
|
|
120
|
+
return Math.ceil(t.length / per);
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/** Where a section sorts in a declared order; unknown names sort last, stably. */
|
|
124
|
+
function rankIn(order, name) {
|
|
125
|
+
const i = order.indexOf(name);
|
|
126
|
+
return i === -1 ? order.length : i;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/** `1 earlier turn` / `14 earlier turns` — the count is the point, so it leads. */
|
|
130
|
+
function turnNote(n) {
|
|
131
|
+
return `… ${n} earlier turn${n === 1 ? "" : "s"} not shown`;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** `… 1,204 characters not shown`. */
|
|
135
|
+
function charNote(n) {
|
|
136
|
+
return `… ${n.toLocaleString("en-US")} character${n === 1 ? "" : "s"} not shown`;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* @typedef {Object} SectionInput
|
|
141
|
+
* @property {string} name one of {@link SECTION_ORDER}, or any name (sorts last)
|
|
142
|
+
* @property {string} [title] heading rendered above the body
|
|
143
|
+
* @property {string} [text] a whole-text section
|
|
144
|
+
* @property {string[]} [turns] a turn-structured section, OLDEST FIRST
|
|
145
|
+
*/
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* @typedef {Object} FittedSection
|
|
149
|
+
* @property {string} name
|
|
150
|
+
* @property {boolean} included false == dropped whole (still rendered as a note)
|
|
151
|
+
* @property {string} text what was kept, drop note included
|
|
152
|
+
* @property {number} keptTurns
|
|
153
|
+
* @property {number} droppedTurns
|
|
154
|
+
* @property {number} droppedChars
|
|
155
|
+
* @property {number} tokens
|
|
156
|
+
*/
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Fit the sections into the tier's budget.
|
|
160
|
+
*
|
|
161
|
+
* Turn-structured sections spend the budget from the NEWEST turn backwards —
|
|
162
|
+
* the turns nearest the message being answered are the ones that explain it
|
|
163
|
+
* (the same rule `lib/org/inbound/hydrate.mjs#renderThread` follows, and for the
|
|
164
|
+
* same reason: clipping the joined string keeps the oldest turns and drops the
|
|
165
|
+
* newest, which is the exact inverse of what context is for).
|
|
166
|
+
*
|
|
167
|
+
* NEVER THROWS. A malformed section is skipped; a missing budget takes the
|
|
168
|
+
* tier's. The prompt path must not be able to fail on its own budgeting.
|
|
169
|
+
*
|
|
170
|
+
* @param {Object} o
|
|
171
|
+
* @param {SectionInput[]} o.sections
|
|
172
|
+
* @param {string} [o.tier] answer | work | plan
|
|
173
|
+
* @param {number} [o.budgetTokens] override the tier's budget
|
|
174
|
+
* @param {number} [o.charsPerToken]
|
|
175
|
+
* @param {(text:string)=>number} [o.estimate] a real tokeniser, when the caller has one
|
|
176
|
+
* @returns {{tier:string, budgetTokens:number, usedTokens:number,
|
|
177
|
+
* sections:FittedSection[], drops:Array<object>, text:string}}
|
|
178
|
+
*/
|
|
179
|
+
export function fitSections(o = {}) {
|
|
180
|
+
const tier = s(o.tier).toLowerCase() || "work";
|
|
181
|
+
const budgetTokens = Number.isFinite(o.budgetTokens) && o.budgetTokens > 0
|
|
182
|
+
? Math.floor(o.budgetTokens)
|
|
183
|
+
: budgetFor(tier);
|
|
184
|
+
const charsPerToken = Number.isFinite(o.charsPerToken) && o.charsPerToken > 0 ? o.charsPerToken : CHARS_PER_TOKEN;
|
|
185
|
+
const tokensOf = typeof o.estimate === "function" ? o.estimate : (t) => estimateTokens(t, charsPerToken);
|
|
186
|
+
|
|
187
|
+
const input = (Array.isArray(o.sections) ? o.sections : [])
|
|
188
|
+
.map((sec, i) => ({ sec, i }))
|
|
189
|
+
.filter(({ sec }) => sec && typeof sec === "object" && s(sec.name))
|
|
190
|
+
.sort((a, b) => rankIn(FILL_PRIORITY, s(a.sec.name)) - rankIn(FILL_PRIORITY, s(b.sec.name)) || a.i - b.i);
|
|
191
|
+
|
|
192
|
+
const out = [];
|
|
193
|
+
const drops = [];
|
|
194
|
+
let used = 0;
|
|
195
|
+
|
|
196
|
+
for (const { sec } of input) {
|
|
197
|
+
const name = s(sec.name);
|
|
198
|
+
const title = s(sec.title);
|
|
199
|
+
const headCost = title ? tokensOf(`${title}\n`) : 0;
|
|
200
|
+
const room = budgetTokens - used - headCost;
|
|
201
|
+
|
|
202
|
+
const fitted = Array.isArray(sec.turns)
|
|
203
|
+
? fitTurns(sec.turns, room, tokensOf)
|
|
204
|
+
: fitText(s(sec.text), room, tokensOf, charsPerToken);
|
|
205
|
+
|
|
206
|
+
// A section with no content at all is simply absent — there is nothing to
|
|
207
|
+
// report the loss of, and a note about an empty section is noise.
|
|
208
|
+
if (fitted.empty) continue;
|
|
209
|
+
|
|
210
|
+
if (!fitted.body && !NEVER_DROPPED_WHOLE.has(name)) {
|
|
211
|
+
// Dropped whole. It still shows: an absence the reader cannot see is the
|
|
212
|
+
// failure this module exists to prevent.
|
|
213
|
+
const note = `… ${name} not shown`;
|
|
214
|
+
const cost = tokensOf(note);
|
|
215
|
+
if (used + cost <= budgetTokens) {
|
|
216
|
+
out.push({ name, included: false, text: note, keptTurns: 0, droppedTurns: fitted.droppedTurns, droppedChars: fitted.droppedChars, tokens: cost });
|
|
217
|
+
used += cost;
|
|
218
|
+
} else {
|
|
219
|
+
out.push({ name, included: false, text: "", keptTurns: 0, droppedTurns: fitted.droppedTurns, droppedChars: fitted.droppedChars, tokens: 0 });
|
|
220
|
+
}
|
|
221
|
+
drops.push({ section: name, turns: fitted.droppedTurns, chars: fitted.droppedChars, note });
|
|
222
|
+
continue;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
// The trigger is the thing being answered: it always renders something,
|
|
226
|
+
// even when the budget cannot pay for it. A prompt with no trigger is not a
|
|
227
|
+
// smaller prompt, it is a broken one.
|
|
228
|
+
let body = fitted.body;
|
|
229
|
+
if (!body && NEVER_DROPPED_WHOLE.has(name)) {
|
|
230
|
+
const minChars = Math.max(1, Math.floor(Math.max(room, 1) * charsPerToken));
|
|
231
|
+
const whole = Array.isArray(sec.turns) ? sec.turns.filter(Boolean).map(s).join("\n") : s(sec.text);
|
|
232
|
+
body = whole.slice(0, minChars);
|
|
233
|
+
const lost = whole.length - body.length;
|
|
234
|
+
if (lost > 0) {
|
|
235
|
+
body += `\n${charNote(lost)}`;
|
|
236
|
+
drops.push({ section: name, turns: 0, chars: lost, note: charNote(lost) });
|
|
237
|
+
}
|
|
238
|
+
} else if (fitted.droppedTurns > 0) {
|
|
239
|
+
drops.push({ section: name, turns: fitted.droppedTurns, chars: 0, note: turnNote(fitted.droppedTurns) });
|
|
240
|
+
} else if (fitted.droppedChars > 0) {
|
|
241
|
+
drops.push({ section: name, turns: 0, chars: fitted.droppedChars, note: charNote(fitted.droppedChars) });
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
const text = title ? `${title}\n${body}` : body;
|
|
245
|
+
const cost = tokensOf(text);
|
|
246
|
+
out.push({
|
|
247
|
+
name,
|
|
248
|
+
included: true,
|
|
249
|
+
text,
|
|
250
|
+
keptTurns: fitted.keptTurns,
|
|
251
|
+
droppedTurns: fitted.droppedTurns,
|
|
252
|
+
droppedChars: fitted.droppedChars,
|
|
253
|
+
tokens: cost,
|
|
254
|
+
});
|
|
255
|
+
used += cost;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
// Filled in priority order, RENDERED in cache order. The two are separate
|
|
259
|
+
// arrays for the reason RENDER_ORDER states, and this is the only place the
|
|
260
|
+
// difference is applied.
|
|
261
|
+
const rendered = out
|
|
262
|
+
.map((sec, i) => ({ sec, i }))
|
|
263
|
+
.sort((a, b) => rankIn(RENDER_ORDER, a.sec.name) - rankIn(RENDER_ORDER, b.sec.name) || a.i - b.i)
|
|
264
|
+
.map(({ sec }) => sec);
|
|
265
|
+
|
|
266
|
+
return {
|
|
267
|
+
tier,
|
|
268
|
+
budgetTokens,
|
|
269
|
+
usedTokens: used,
|
|
270
|
+
sections: rendered,
|
|
271
|
+
drops,
|
|
272
|
+
text: rendered.map((sec) => sec.text).filter(Boolean).join("\n\n"),
|
|
273
|
+
};
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
/** Keep the newest turns that fit; report the rest as a visible note. */
|
|
277
|
+
function fitTurns(turns, room, tokensOf) {
|
|
278
|
+
const rows = (Array.isArray(turns) ? turns : []).filter((t) => s(t).trim()).map((t) => s(t));
|
|
279
|
+
if (rows.length === 0) return { empty: true, body: "", keptTurns: 0, droppedTurns: 0, droppedChars: 0 };
|
|
280
|
+
if (room <= 0) return { empty: false, body: "", keptTurns: 0, droppedTurns: rows.length, droppedChars: 0 };
|
|
281
|
+
|
|
282
|
+
const kept = [];
|
|
283
|
+
let used = 0;
|
|
284
|
+
for (let i = rows.length - 1; i >= 0; i -= 1) {
|
|
285
|
+
const cost = tokensOf(rows[i]) + (kept.length ? 1 : 0);
|
|
286
|
+
// The note itself has to fit, or the drop would be the silent kind.
|
|
287
|
+
const reserve = i > 0 ? tokensOf(turnNote(i)) + 1 : 0;
|
|
288
|
+
if (used + cost + reserve > room) break;
|
|
289
|
+
kept.unshift(rows[i]);
|
|
290
|
+
used += cost;
|
|
291
|
+
}
|
|
292
|
+
const dropped = rows.length - kept.length;
|
|
293
|
+
if (kept.length === 0) return { empty: false, body: "", keptTurns: 0, droppedTurns: dropped, droppedChars: 0 };
|
|
294
|
+
const body = dropped > 0 ? [turnNote(dropped), ...kept].join("\n") : kept.join("\n");
|
|
295
|
+
return { empty: false, body, keptTurns: kept.length, droppedTurns: dropped, droppedChars: 0 };
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
/** Keep the head that fits; report the tail as a visible note. */
|
|
299
|
+
function fitText(text, room, tokensOf, charsPerToken) {
|
|
300
|
+
const t = s(text).trim();
|
|
301
|
+
if (!t) return { empty: true, body: "", keptTurns: 0, droppedTurns: 0, droppedChars: 0 };
|
|
302
|
+
if (room <= 0) return { empty: false, body: "", keptTurns: 0, droppedTurns: 0, droppedChars: t.length };
|
|
303
|
+
if (tokensOf(t) <= room) return { empty: false, body: t, keptTurns: 1, droppedTurns: 0, droppedChars: 0 };
|
|
304
|
+
|
|
305
|
+
// Reserve room for the note, then take the largest head that still fits.
|
|
306
|
+
let chars = Math.max(0, Math.floor(room * charsPerToken));
|
|
307
|
+
let head = t.slice(0, chars);
|
|
308
|
+
let note = charNote(t.length - head.length);
|
|
309
|
+
while (head.length > 0 && tokensOf(`${head}\n${note}`) > room) {
|
|
310
|
+
chars = Math.floor(chars * 0.8);
|
|
311
|
+
head = t.slice(0, chars);
|
|
312
|
+
note = charNote(t.length - head.length);
|
|
313
|
+
}
|
|
314
|
+
if (!head) return { empty: false, body: "", keptTurns: 0, droppedTurns: 0, droppedChars: t.length };
|
|
315
|
+
return { empty: false, body: `${head}\n${note}`, keptTurns: 1, droppedTurns: 0, droppedChars: t.length - head.length };
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
export default {
|
|
319
|
+
TIER_BUDGETS,
|
|
320
|
+
FILL_PRIORITY,
|
|
321
|
+
RENDER_ORDER,
|
|
322
|
+
SECTION_ORDER,
|
|
323
|
+
CHARS_PER_TOKEN,
|
|
324
|
+
budgetFor,
|
|
325
|
+
estimateTokens,
|
|
326
|
+
fitSections,
|
|
327
|
+
};
|
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* budget.test.mjs — the declared token budget (spec §5.4, SDK half).
|
|
3
|
+
*
|
|
4
|
+
* Three properties are under test throughout:
|
|
5
|
+
* 1. the budget is DECLARED per tier, not discovered by a char clamp;
|
|
6
|
+
* 2. sections are filled in PRIORITY order, so the trigger survives a budget
|
|
7
|
+
* that the backlog cannot, and RENDERED in cache order, which is a
|
|
8
|
+
* different order for a different reason; and
|
|
9
|
+
* 3. every drop is VISIBLE in the rendered text. A silent drop is the bug —
|
|
10
|
+
* the reader cannot tell an absent turn from a turn that never happened.
|
|
11
|
+
*
|
|
12
|
+
* Run: node --test lib/context/budget.test.mjs
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
"use strict";
|
|
16
|
+
|
|
17
|
+
import { test } from "node:test";
|
|
18
|
+
import assert from "node:assert/strict";
|
|
19
|
+
|
|
20
|
+
import {
|
|
21
|
+
TIER_BUDGETS,
|
|
22
|
+
SECTION_ORDER,
|
|
23
|
+
FILL_PRIORITY,
|
|
24
|
+
RENDER_ORDER,
|
|
25
|
+
CHARS_PER_TOKEN,
|
|
26
|
+
budgetFor,
|
|
27
|
+
estimateTokens,
|
|
28
|
+
fitSections,
|
|
29
|
+
} from "./budget.mjs";
|
|
30
|
+
|
|
31
|
+
const turns = (n, text = "x".repeat(40)) => Array.from({ length: n }, (_, i) => `s${i}: ${text}`);
|
|
32
|
+
|
|
33
|
+
// ---------------------------------------------------------------------------
|
|
34
|
+
// the declared budget
|
|
35
|
+
// ---------------------------------------------------------------------------
|
|
36
|
+
|
|
37
|
+
test("each tier declares its own budget, and the table is frozen", () => {
|
|
38
|
+
assert.equal(TIER_BUDGETS.answer, 6000);
|
|
39
|
+
assert.equal(TIER_BUDGETS.work, 12000);
|
|
40
|
+
assert.equal(TIER_BUDGETS.plan, 20000);
|
|
41
|
+
assert.equal(Object.isFrozen(TIER_BUDGETS), true);
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
test("an unknown tier takes the WORK budget rather than an unbounded one", () => {
|
|
45
|
+
assert.equal(budgetFor("plan"), 20000);
|
|
46
|
+
assert.equal(budgetFor("nonsense"), TIER_BUDGETS.work);
|
|
47
|
+
assert.equal(budgetFor(undefined), TIER_BUDGETS.work);
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
test("estimateTokens is the declared chars-per-token ratio, rounded up", () => {
|
|
51
|
+
assert.equal(estimateTokens(""), 0);
|
|
52
|
+
assert.equal(estimateTokens("a".repeat(CHARS_PER_TOKEN)), 1);
|
|
53
|
+
assert.equal(estimateTokens("a".repeat(CHARS_PER_TOKEN + 1)), 2);
|
|
54
|
+
assert.equal(estimateTokens(null), 0);
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
// ---------------------------------------------------------------------------
|
|
58
|
+
// fitting
|
|
59
|
+
// ---------------------------------------------------------------------------
|
|
60
|
+
|
|
61
|
+
test("everything that fits is kept, verbatim, with no drop notes", () => {
|
|
62
|
+
const out = fitSections({
|
|
63
|
+
tier: "plan",
|
|
64
|
+
sections: [
|
|
65
|
+
{ name: "trigger", text: "fix the invoice total" },
|
|
66
|
+
{ name: "thread", turns: turns(3) },
|
|
67
|
+
],
|
|
68
|
+
});
|
|
69
|
+
assert.equal(out.drops.length, 0);
|
|
70
|
+
assert.match(out.text, /fix the invoice total/);
|
|
71
|
+
assert.doesNotMatch(out.text, /not shown/);
|
|
72
|
+
assert.ok(out.usedTokens <= out.budgetTokens);
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
test("an overflowing thread keeps the NEWEST turns and says how many it dropped", () => {
|
|
76
|
+
const out = fitSections({
|
|
77
|
+
tier: "answer",
|
|
78
|
+
budgetTokens: 60, // ~240 chars
|
|
79
|
+
sections: [
|
|
80
|
+
{ name: "trigger", text: "which one?" },
|
|
81
|
+
{ name: "thread", turns: turns(20, "y".repeat(60)) },
|
|
82
|
+
],
|
|
83
|
+
});
|
|
84
|
+
const thread = out.sections.find((s) => s.name === "thread");
|
|
85
|
+
assert.ok(thread.keptTurns > 0, "at least one turn survives");
|
|
86
|
+
assert.ok(thread.droppedTurns > 0, "the rest were dropped");
|
|
87
|
+
assert.equal(thread.keptTurns + thread.droppedTurns, 20);
|
|
88
|
+
// The newest turn is the one nearest the message being answered.
|
|
89
|
+
assert.match(out.text, /s19:/);
|
|
90
|
+
assert.doesNotMatch(out.text, /s0:/);
|
|
91
|
+
assert.match(out.text, new RegExp(`… ${thread.droppedTurns} earlier turns not shown`));
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
test("the drop note is singular for exactly one dropped turn", () => {
|
|
95
|
+
// Two turns, and room for the newest one plus its drop note but not for both.
|
|
96
|
+
const out = fitSections({
|
|
97
|
+
tier: "answer",
|
|
98
|
+
budgetTokens: 60,
|
|
99
|
+
sections: [{ name: "thread", turns: [`s0: ${"a".repeat(200)}`, `s1: ${"b".repeat(200)}`] }],
|
|
100
|
+
});
|
|
101
|
+
const thread = out.sections.find((s) => s.name === "thread");
|
|
102
|
+
assert.equal(thread.keptTurns, 1);
|
|
103
|
+
assert.equal(thread.droppedTurns, 1);
|
|
104
|
+
assert.match(out.text, /… 1 earlier turn not shown/);
|
|
105
|
+
assert.doesNotMatch(out.text, /earlier turns not shown/);
|
|
106
|
+
});
|
|
107
|
+
|
|
108
|
+
test("sections fill in priority order — the trigger outranks the backlog", () => {
|
|
109
|
+
const out = fitSections({
|
|
110
|
+
tier: "answer",
|
|
111
|
+
budgetTokens: 12,
|
|
112
|
+
sections: [
|
|
113
|
+
{ name: "backlog", turns: turns(6, "b".repeat(80)) },
|
|
114
|
+
{ name: "trigger", text: "the one question that matters" },
|
|
115
|
+
],
|
|
116
|
+
});
|
|
117
|
+
assert.match(out.text, /the one question that matters/);
|
|
118
|
+
const backlog = out.sections.find((s) => s.name === "backlog");
|
|
119
|
+
assert.equal(backlog.included, false);
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
test("a section dropped WHOLE is still visible — never silently absent", () => {
|
|
123
|
+
const out = fitSections({
|
|
124
|
+
tier: "answer",
|
|
125
|
+
budgetTokens: 20,
|
|
126
|
+
sections: [
|
|
127
|
+
{ name: "trigger", text: "t".repeat(60) },
|
|
128
|
+
{ name: "memory", turns: turns(4, "m".repeat(100)) },
|
|
129
|
+
],
|
|
130
|
+
});
|
|
131
|
+
const memory = out.sections.find((s) => s.name === "memory");
|
|
132
|
+
assert.equal(memory.included, false);
|
|
133
|
+
assert.match(out.text, /… memory not shown/);
|
|
134
|
+
assert.equal(out.drops.some((d) => d.section === "memory"), true);
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
test("a long text section is clipped and says how many characters went", () => {
|
|
138
|
+
const body = "z".repeat(4000);
|
|
139
|
+
const out = fitSections({
|
|
140
|
+
tier: "answer",
|
|
141
|
+
budgetTokens: 100,
|
|
142
|
+
sections: [{ name: "entity", title: "Document", text: body }],
|
|
143
|
+
});
|
|
144
|
+
const entity = out.sections.find((s) => s.name === "entity");
|
|
145
|
+
assert.ok(entity.droppedChars > 0);
|
|
146
|
+
assert.match(out.text, /characters not shown/);
|
|
147
|
+
assert.ok(out.usedTokens <= out.budgetTokens);
|
|
148
|
+
});
|
|
149
|
+
|
|
150
|
+
test("the trigger is never dropped whole — it is the thing being answered", () => {
|
|
151
|
+
const out = fitSections({
|
|
152
|
+
tier: "answer",
|
|
153
|
+
budgetTokens: 5,
|
|
154
|
+
sections: [{ name: "trigger", text: "q".repeat(2000) }],
|
|
155
|
+
});
|
|
156
|
+
const trigger = out.sections.find((s) => s.name === "trigger");
|
|
157
|
+
assert.equal(trigger.included, true);
|
|
158
|
+
assert.ok(trigger.text.length > 0);
|
|
159
|
+
});
|
|
160
|
+
|
|
161
|
+
test("rendered order is CACHE order — stable first, the trigger last", () => {
|
|
162
|
+
// Fill priority and render order are two different questions. The trigger
|
|
163
|
+
// WINS the budget (it is the thing being answered) and RENDERS last (it is
|
|
164
|
+
// the only section that changes on every inbound). A block that opened with
|
|
165
|
+
// the trigger would have a zero-length stable prefix, so WP-4's cache_control
|
|
166
|
+
// breakpoint would sit after volatile content and every cache read would miss
|
|
167
|
+
// silently — design §4, "any byte change before a breakpoint invalidates
|
|
168
|
+
// everything after it".
|
|
169
|
+
const out = fitSections({
|
|
170
|
+
tier: "plan",
|
|
171
|
+
sections: [
|
|
172
|
+
{ name: "memory", turns: ["m: one"] },
|
|
173
|
+
{ name: "trigger", text: "the ask" },
|
|
174
|
+
{ name: "thread", turns: ["t: two"] },
|
|
175
|
+
{ name: "entity", text: "the card" },
|
|
176
|
+
],
|
|
177
|
+
});
|
|
178
|
+
assert.deepEqual(out.sections.map((s) => s.name), ["entity", "memory", "thread", "trigger"]);
|
|
179
|
+
assert.ok(out.text.indexOf("the card") < out.text.indexOf("m: one"));
|
|
180
|
+
assert.ok(out.text.indexOf("m: one") < out.text.indexOf("t: two"));
|
|
181
|
+
assert.ok(out.text.indexOf("t: two") < out.text.indexOf("the ask"));
|
|
182
|
+
assert.ok(out.text.endsWith("the ask"), "the volatile section is the tail, so everything before it can be cached");
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
test("FILL priority still decides who is dropped — the trigger wins the budget", () => {
|
|
186
|
+
// Same four sections, a budget only one of them fits. Render order must not
|
|
187
|
+
// become a fill order by accident: `entity` renders first but `trigger` is
|
|
188
|
+
// what survives.
|
|
189
|
+
const out = fitSections({
|
|
190
|
+
budgetTokens: 8,
|
|
191
|
+
sections: [
|
|
192
|
+
{ name: "entity", text: "E".repeat(400) },
|
|
193
|
+
{ name: "backlog", text: "B".repeat(400) },
|
|
194
|
+
{ name: "trigger", text: "the ask itself" },
|
|
195
|
+
],
|
|
196
|
+
});
|
|
197
|
+
const byName = Object.fromEntries(out.sections.map((s) => [s.name, s]));
|
|
198
|
+
assert.equal(byName.trigger.included, true, "the trigger is filled first, whatever it renders after");
|
|
199
|
+
assert.equal(byName.trigger.droppedChars, 0, "…and it is filled WHOLE");
|
|
200
|
+
assert.match(out.text, /the ask itself/);
|
|
201
|
+
assert.equal(byName.backlog.included, false, "the last section in fill priority is the first to go");
|
|
202
|
+
assert.ok(byName.entity.droppedChars > 0, "and the higher-priority one is only trimmed");
|
|
203
|
+
assert.equal(out.text.endsWith("the ask itself"), true, "render order is unchanged by who was dropped");
|
|
204
|
+
});
|
|
205
|
+
|
|
206
|
+
test("FILL_PRIORITY and RENDER_ORDER name the same sections, and are not the same array", () => {
|
|
207
|
+
assert.deepEqual([...FILL_PRIORITY].sort(), [...RENDER_ORDER].sort(),
|
|
208
|
+
"a section that can be filled must be renderable, and vice versa");
|
|
209
|
+
assert.notDeepEqual([...FILL_PRIORITY], [...RENDER_ORDER]);
|
|
210
|
+
assert.equal(FILL_PRIORITY[0], "trigger", "the trigger wins the budget");
|
|
211
|
+
assert.equal(RENDER_ORDER[RENDER_ORDER.length - 1], "trigger", "…and renders last, so the prefix is cacheable");
|
|
212
|
+
});
|
|
213
|
+
|
|
214
|
+
test("an unknown section name sorts after every declared one, keeping its order", () => {
|
|
215
|
+
const out = fitSections({
|
|
216
|
+
tier: "plan",
|
|
217
|
+
sections: [
|
|
218
|
+
{ name: "weather", text: "w" },
|
|
219
|
+
{ name: "trigger", text: "t" },
|
|
220
|
+
{ name: "tides", text: "d" },
|
|
221
|
+
],
|
|
222
|
+
});
|
|
223
|
+
assert.deepEqual(out.sections.map((s) => s.name), ["trigger", "weather", "tides"]);
|
|
224
|
+
assert.equal(SECTION_ORDER.includes("trigger"), true);
|
|
225
|
+
assert.equal(SECTION_ORDER.includes("weather"), false);
|
|
226
|
+
});
|
|
227
|
+
|
|
228
|
+
// ---------------------------------------------------------------------------
|
|
229
|
+
// purity
|
|
230
|
+
// ---------------------------------------------------------------------------
|
|
231
|
+
|
|
232
|
+
test("pure: same input, same output, and the caller's arrays are untouched", () => {
|
|
233
|
+
const input = { tier: "work", sections: [{ name: "thread", turns: turns(50) }] };
|
|
234
|
+
const before = JSON.stringify(input);
|
|
235
|
+
const a = fitSections(input);
|
|
236
|
+
const b = fitSections(input);
|
|
237
|
+
assert.deepEqual(a, b);
|
|
238
|
+
assert.equal(JSON.stringify(input), before, "fitSections mutated its argument");
|
|
239
|
+
});
|
|
240
|
+
|
|
241
|
+
test("no sections is an empty block, not a throw", () => {
|
|
242
|
+
const out = fitSections({ tier: "answer", sections: [] });
|
|
243
|
+
assert.equal(out.text, "");
|
|
244
|
+
assert.equal(out.usedTokens, 0);
|
|
245
|
+
assert.deepEqual(out.drops, []);
|
|
246
|
+
});
|
|
247
|
+
|
|
248
|
+
test("a malformed section is ignored rather than taking the prompt down", () => {
|
|
249
|
+
const out = fitSections({ tier: "answer", sections: [null, { text: "no name" }, { name: "trigger", text: "ok" }] });
|
|
250
|
+
assert.equal(out.sections.length, 1);
|
|
251
|
+
assert.match(out.text, /ok/);
|
|
252
|
+
});
|