@oh-my-pi/pi-agent-core 17.3.8 → 17.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/dist/types/agent.d.ts +8 -1
- package/dist/types/compaction/branch-summarization.d.ts +2 -1
- package/dist/types/compaction/compaction.d.ts +24 -17
- package/dist/types/compaction/index.d.ts +1 -0
- package/dist/types/compaction/message-cache.d.ts +6 -4
- package/dist/types/compaction/messages.d.ts +17 -1
- package/dist/types/compaction/openai.d.ts +2 -1
- package/dist/types/compaction/pruning.d.ts +3 -2
- package/dist/types/compaction/shake.d.ts +2 -1
- package/dist/types/compaction/transcript-tokens.d.ts +76 -0
- package/dist/types/tokenizer.d.ts +78 -2
- package/dist/types/types.d.ts +1 -1
- package/package.json +8 -8
- package/src/agent.ts +25 -4
- package/src/compaction/branch-summarization.ts +15 -8
- package/src/compaction/compaction.ts +43 -152
- package/src/compaction/index.ts +1 -0
- package/src/compaction/message-cache.ts +33 -34
- package/src/compaction/messages.ts +21 -5
- package/src/compaction/openai.ts +46 -18
- package/src/compaction/pruning.ts +25 -13
- package/src/compaction/shake.ts +18 -16
- package/src/compaction/transcript-tokens.ts +111 -0
- package/src/tokenizer.ts +273 -18
- package/src/types.ts +1 -1
|
@@ -3,8 +3,8 @@
|
|
|
3
3
|
*/
|
|
4
4
|
|
|
5
5
|
import type { ToolResultMessage } from "@oh-my-pi/pi-ai";
|
|
6
|
+
import type { Tokenizer } from "../tokenizer";
|
|
6
7
|
import type { AgentMessage, AgentToolCall } from "../types";
|
|
7
|
-
import { estimateTokens } from "./compaction";
|
|
8
8
|
import type { SessionEntry, SessionMessageEntry } from "./entries";
|
|
9
9
|
import { invalidateMessageCache } from "./message-cache";
|
|
10
10
|
import {
|
|
@@ -140,13 +140,13 @@ function estimatePrunedSavings(tokens: number, notice: string): number {
|
|
|
140
140
|
* (cacheWrite premium) if that entry is mutated in place. Used to keep prune
|
|
141
141
|
* mutations inside the cheap-to-recache tail.
|
|
142
142
|
*/
|
|
143
|
-
function computeMessageSuffixTokens(entries: readonly SessionEntry[]): number[] {
|
|
143
|
+
function computeMessageSuffixTokens(entries: readonly SessionEntry[], tokenizer: Tokenizer): number[] {
|
|
144
144
|
const suffix = new Array<number>(entries.length);
|
|
145
145
|
let accumulated = 0;
|
|
146
146
|
for (let i = entries.length - 1; i >= 0; i--) {
|
|
147
147
|
suffix[i] = accumulated;
|
|
148
148
|
const entry = entries[i];
|
|
149
|
-
if (entry.type === "message") accumulated +=
|
|
149
|
+
if (entry.type === "message") accumulated += tokenizer.countMessage(entry.message as AgentMessage);
|
|
150
150
|
}
|
|
151
151
|
return suffix;
|
|
152
152
|
}
|
|
@@ -181,6 +181,7 @@ interface SupersedeCandidate {
|
|
|
181
181
|
*/
|
|
182
182
|
function collectSupersededResults(
|
|
183
183
|
entries: readonly SessionEntry[],
|
|
184
|
+
tokenizer: Tokenizer,
|
|
184
185
|
toolCallsById: ReadonlyMap<string, AgentToolCall>,
|
|
185
186
|
supersedeKey: SupersedeKeyFn,
|
|
186
187
|
protectedTools: readonly ProtectedToolMatcher[],
|
|
@@ -204,7 +205,7 @@ function collectSupersededResults(
|
|
|
204
205
|
entry: entry as SessionMessageEntry,
|
|
205
206
|
message,
|
|
206
207
|
index: i,
|
|
207
|
-
tokens:
|
|
208
|
+
tokens: tokenizer.countMessage(message as AgentMessage),
|
|
208
209
|
notice: SUPERSEDED_NOTICE,
|
|
209
210
|
});
|
|
210
211
|
}
|
|
@@ -219,6 +220,7 @@ function collectSupersededResults(
|
|
|
219
220
|
*/
|
|
220
221
|
function collectUselessResults(
|
|
221
222
|
entries: readonly SessionEntry[],
|
|
223
|
+
tokenizer: Tokenizer,
|
|
222
224
|
toolCallsById: ReadonlyMap<string, AgentToolCall>,
|
|
223
225
|
protectedTools: readonly ProtectedToolMatcher[],
|
|
224
226
|
exclude: ReadonlySet<ToolResultMessage>,
|
|
@@ -230,7 +232,7 @@ function collectUselessResults(
|
|
|
230
232
|
if (message?.useless !== true || message.prunedAt !== undefined || message.isError === true) continue;
|
|
231
233
|
if (exclude.has(message)) continue;
|
|
232
234
|
if (isProtectedToolResult(message, toolCallsById.get(message.toolCallId), protectedTools)) continue;
|
|
233
|
-
const tokens =
|
|
235
|
+
const tokens = tokenizer.countMessage(message as AgentMessage);
|
|
234
236
|
if (estimatePrunedSavings(tokens, USELESS_NOTICE) <= 0) continue;
|
|
235
237
|
candidates.push({ entry: entry as SessionMessageEntry, message, index: i, tokens, notice: USELESS_NOTICE });
|
|
236
238
|
}
|
|
@@ -246,14 +248,18 @@ function collectUselessResults(
|
|
|
246
248
|
* the provider cache is cold anyway (then all still-sent candidates flush).
|
|
247
249
|
* Never mutates entries before `keepBoundaryId` (summarized away — not sent).
|
|
248
250
|
*/
|
|
249
|
-
export function pruneSupersededToolResults(
|
|
251
|
+
export function pruneSupersededToolResults(
|
|
252
|
+
entries: SessionEntry[],
|
|
253
|
+
tokenizer: Tokenizer,
|
|
254
|
+
config: SupersedePruneConfig,
|
|
255
|
+
): PruneResult {
|
|
250
256
|
const toolCallsById = collectToolCallsById(entries);
|
|
251
257
|
const candidates = config.supersedeKey
|
|
252
|
-
? collectSupersededResults(entries, toolCallsById, config.supersedeKey, config.protectedTools)
|
|
258
|
+
? collectSupersededResults(entries, tokenizer, toolCallsById, config.supersedeKey, config.protectedTools)
|
|
253
259
|
: [];
|
|
254
260
|
if (config.pruneUseless) {
|
|
255
261
|
const exclude = new Set(candidates.map(candidate => candidate.message));
|
|
256
|
-
candidates.push(...collectUselessResults(entries, toolCallsById, config.protectedTools, exclude));
|
|
262
|
+
candidates.push(...collectUselessResults(entries, tokenizer, toolCallsById, config.protectedTools, exclude));
|
|
257
263
|
candidates.sort((a, b) => a.index - b.index);
|
|
258
264
|
}
|
|
259
265
|
if (candidates.length === 0) return { prunedCount: 0, tokensSaved: 0 };
|
|
@@ -284,7 +290,7 @@ export function pruneSupersededToolResults(entries: SessionEntry[], config: Supe
|
|
|
284
290
|
// Mutating a candidate re-writes its suffix in the warm cache, so prune only
|
|
285
291
|
// when that suffix is small (cheap-to-recache tail) and the candidate sits
|
|
286
292
|
// at/after the compaction boundary.
|
|
287
|
-
const suffixTokens = computeMessageSuffixTokens(entries);
|
|
293
|
+
const suffixTokens = computeMessageSuffixTokens(entries, tokenizer);
|
|
288
294
|
toPrune = candidates.filter(
|
|
289
295
|
candidate => candidate.index >= boundaryIndex && suffixTokens[candidate.index] <= suffixTokenLimit,
|
|
290
296
|
);
|
|
@@ -302,7 +308,11 @@ export function pruneSupersededToolResults(entries: SessionEntry[], config: Supe
|
|
|
302
308
|
return { prunedCount: toPrune.length, tokensSaved };
|
|
303
309
|
}
|
|
304
310
|
|
|
305
|
-
export function pruneToolOutputs(
|
|
311
|
+
export function pruneToolOutputs(
|
|
312
|
+
entries: SessionEntry[],
|
|
313
|
+
tokenizer: Tokenizer,
|
|
314
|
+
config: PruneConfig = DEFAULT_PRUNE_CONFIG,
|
|
315
|
+
): PruneResult {
|
|
306
316
|
let accumulatedTokens = 0;
|
|
307
317
|
let tokensSaved = 0;
|
|
308
318
|
let prunedCount = 0;
|
|
@@ -311,7 +321,7 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig =
|
|
|
311
321
|
const toolCallsById = collectToolCallsById(entries);
|
|
312
322
|
const supersededMessages = config.supersedeKey
|
|
313
323
|
? new Set(
|
|
314
|
-
collectSupersededResults(entries, toolCallsById, config.supersedeKey, config.protectedTools).map(
|
|
324
|
+
collectSupersededResults(entries, tokenizer, toolCallsById, config.supersedeKey, config.protectedTools).map(
|
|
315
325
|
candidate => candidate.message,
|
|
316
326
|
),
|
|
317
327
|
)
|
|
@@ -321,6 +331,7 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig =
|
|
|
321
331
|
? new Set(
|
|
322
332
|
collectUselessResults(
|
|
323
333
|
entries,
|
|
334
|
+
tokenizer,
|
|
324
335
|
toolCallsById,
|
|
325
336
|
config.protectedTools,
|
|
326
337
|
supersededMessages ?? new Set(),
|
|
@@ -331,14 +342,15 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig =
|
|
|
331
342
|
const boundaryIndex = resolveBoundaryIndex(entries, config.keepBoundaryId);
|
|
332
343
|
const cacheWarmSuffixTokens = config.cacheWarmSuffixTokens;
|
|
333
344
|
// All-message suffix per index, only when the cache guard is armed.
|
|
334
|
-
const messageSuffix =
|
|
345
|
+
const messageSuffix =
|
|
346
|
+
cacheWarmSuffixTokens === undefined ? undefined : computeMessageSuffixTokens(entries, tokenizer);
|
|
335
347
|
|
|
336
348
|
for (let i = entries.length - 1; i >= 0; i--) {
|
|
337
349
|
const entry = entries[i];
|
|
338
350
|
const message = getToolResultMessage(entry);
|
|
339
351
|
if (!message) continue;
|
|
340
352
|
|
|
341
|
-
const tokens =
|
|
353
|
+
const tokens = tokenizer.countMessage(message as AgentMessage);
|
|
342
354
|
const isProtected = isProtectedToolResult(message, toolCallsById.get(message.toolCallId), config.protectedTools);
|
|
343
355
|
|
|
344
356
|
if (message.prunedAt !== undefined) {
|
package/src/compaction/shake.ts
CHANGED
|
@@ -11,9 +11,8 @@
|
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
13
|
import type { TextContent, ToolResultMessage } from "@oh-my-pi/pi-ai";
|
|
14
|
-
import {
|
|
14
|
+
import type { Tokenizer } from "../tokenizer";
|
|
15
15
|
import type { AgentMessage } from "../types";
|
|
16
|
-
import { estimateTokens } from "./compaction";
|
|
17
16
|
import type { CustomMessageEntry, SessionEntry, SessionMessageEntry } from "./entries";
|
|
18
17
|
import { invalidateMessageCache } from "./message-cache";
|
|
19
18
|
import {
|
|
@@ -123,15 +122,15 @@ function toolResultText(message: ToolResultMessage): string {
|
|
|
123
122
|
}
|
|
124
123
|
|
|
125
124
|
/** Estimate the token contribution of an entry for the protect-recent window. */
|
|
126
|
-
function entryTokens(entry: SessionEntry): number {
|
|
125
|
+
function entryTokens(entry: SessionEntry, tokenizer: Tokenizer): number {
|
|
127
126
|
if (entry.type === "message") {
|
|
128
|
-
return
|
|
127
|
+
return tokenizer.countMessage(entry.message);
|
|
129
128
|
}
|
|
130
129
|
if (entry.type === "custom_message") {
|
|
131
130
|
const content = entry.content;
|
|
132
|
-
if (typeof content === "string") return content.length === 0 ? 0 : countTokens(content);
|
|
131
|
+
if (typeof content === "string") return content.length === 0 ? 0 : tokenizer.countTokens(content);
|
|
133
132
|
const fragments = content.filter((block): block is TextContent => block.type === "text").map(block => block.text);
|
|
134
|
-
return fragments.length === 0 ? 0 : countTokens(fragments);
|
|
133
|
+
return fragments.length === 0 ? 0 : tokenizer.countTokens(fragments);
|
|
135
134
|
}
|
|
136
135
|
return 0;
|
|
137
136
|
}
|
|
@@ -222,6 +221,7 @@ function pushBlockRegions(
|
|
|
222
221
|
entry: SessionMessageEntry | CustomMessageEntry,
|
|
223
222
|
blockIndex: number,
|
|
224
223
|
text: string,
|
|
224
|
+
tokenizer: Tokenizer,
|
|
225
225
|
config: ShakeConfig,
|
|
226
226
|
label: string,
|
|
227
227
|
out: ShakeRegion[],
|
|
@@ -229,7 +229,7 @@ function pushBlockRegions(
|
|
|
229
229
|
for (const range of scanTextForBlockRanges(text)) {
|
|
230
230
|
const slice = text.slice(range.start, range.end);
|
|
231
231
|
if (slice.length === 0) continue;
|
|
232
|
-
const tokens = countTokens(slice);
|
|
232
|
+
const tokens = tokenizer.countTokens(slice);
|
|
233
233
|
if (tokens < config.fenceMinTokens) continue;
|
|
234
234
|
out.push({
|
|
235
235
|
kind: "block",
|
|
@@ -246,6 +246,7 @@ function pushBlockRegions(
|
|
|
246
246
|
|
|
247
247
|
function collectBlockRegions(
|
|
248
248
|
entry: SessionMessageEntry | CustomMessageEntry,
|
|
249
|
+
tokenizer: Tokenizer,
|
|
249
250
|
config: ShakeConfig,
|
|
250
251
|
out: ShakeRegion[],
|
|
251
252
|
): void {
|
|
@@ -254,34 +255,35 @@ function collectBlockRegions(
|
|
|
254
255
|
if (message.role === "assistant") {
|
|
255
256
|
for (let bi = 0; bi < message.content.length; bi++) {
|
|
256
257
|
const block = message.content[bi];
|
|
257
|
-
if (block.type === "text") pushBlockRegions(entry, bi, block.text, config, "assistant", out);
|
|
258
|
+
if (block.type === "text") pushBlockRegions(entry, bi, block.text, tokenizer, config, "assistant", out);
|
|
258
259
|
}
|
|
259
260
|
return;
|
|
260
261
|
}
|
|
261
262
|
if (message.role === "user" || message.role === "developer") {
|
|
262
|
-
scanContentBlocks(entry, message.content, config, message.role, out);
|
|
263
|
+
scanContentBlocks(entry, message.content, tokenizer, config, message.role, out);
|
|
263
264
|
}
|
|
264
265
|
return;
|
|
265
266
|
}
|
|
266
267
|
// custom_message
|
|
267
|
-
scanContentBlocks(entry, entry.content, config, entry.customType, out);
|
|
268
|
+
scanContentBlocks(entry, entry.content, tokenizer, config, entry.customType, out);
|
|
268
269
|
}
|
|
269
270
|
|
|
270
271
|
function scanContentBlocks(
|
|
271
272
|
entry: SessionMessageEntry | CustomMessageEntry,
|
|
272
273
|
content: string | Array<{ type: string; text?: string }>,
|
|
274
|
+
tokenizer: Tokenizer,
|
|
273
275
|
config: ShakeConfig,
|
|
274
276
|
label: string,
|
|
275
277
|
out: ShakeRegion[],
|
|
276
278
|
): void {
|
|
277
279
|
if (typeof content === "string") {
|
|
278
|
-
pushBlockRegions(entry, -1, content, config, label, out);
|
|
280
|
+
pushBlockRegions(entry, -1, content, tokenizer, config, label, out);
|
|
279
281
|
return;
|
|
280
282
|
}
|
|
281
283
|
for (let bi = 0; bi < content.length; bi++) {
|
|
282
284
|
const block = content[bi];
|
|
283
285
|
if (block.type === "text" && typeof block.text === "string") {
|
|
284
|
-
pushBlockRegions(entry, bi, block.text, config, label, out);
|
|
286
|
+
pushBlockRegions(entry, bi, block.text, tokenizer, config, label, out);
|
|
285
287
|
}
|
|
286
288
|
}
|
|
287
289
|
}
|
|
@@ -300,7 +302,7 @@ function scanContentBlocks(
|
|
|
300
302
|
* and regions never span a message boundary. When the combined estimated
|
|
301
303
|
* savings is below `minSavings`, returns `[]` (no-op).
|
|
302
304
|
*/
|
|
303
|
-
export function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig): ShakeRegion[] {
|
|
305
|
+
export function collectShakeRegions(entries: SessionEntry[], tokenizer: Tokenizer, config: ShakeConfig): ShakeRegion[] {
|
|
304
306
|
const n = entries.length;
|
|
305
307
|
if (n === 0) return [];
|
|
306
308
|
|
|
@@ -309,7 +311,7 @@ export function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig
|
|
|
309
311
|
let acc = 0;
|
|
310
312
|
for (let i = n - 1; i >= 0; i--) {
|
|
311
313
|
accumulatedAfter[i] = acc;
|
|
312
|
-
acc += entryTokens(entries[i]);
|
|
314
|
+
acc += entryTokens(entries[i], tokenizer);
|
|
313
315
|
}
|
|
314
316
|
|
|
315
317
|
const toolCallsById = collectToolCallsById(entries);
|
|
@@ -342,7 +344,7 @@ export function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig
|
|
|
342
344
|
regions.push({
|
|
343
345
|
kind: "toolResult",
|
|
344
346
|
entry: entry as SessionMessageEntry,
|
|
345
|
-
tokens:
|
|
347
|
+
tokens: tokenizer.countMessage(toolResult as AgentMessage),
|
|
346
348
|
originalText: text,
|
|
347
349
|
label: toolResult.toolName,
|
|
348
350
|
});
|
|
@@ -350,7 +352,7 @@ export function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig
|
|
|
350
352
|
}
|
|
351
353
|
|
|
352
354
|
if (entry.type === "message" || entry.type === "custom_message") {
|
|
353
|
-
collectBlockRegions(entry as SessionMessageEntry | CustomMessageEntry, config, regions);
|
|
355
|
+
collectBlockRegions(entry as SessionMessageEntry | CustomMessageEntry, tokenizer, config, regions);
|
|
354
356
|
}
|
|
355
357
|
}
|
|
356
358
|
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Provider-anchored transcript token accounting.
|
|
3
|
+
*
|
|
4
|
+
* Local tokenization is the expensive way to answer "how big is this
|
|
5
|
+
* conversation?" — and usually the wrong one, because the provider already
|
|
6
|
+
* answered it. Every settled assistant turn carries `usage` covering the exact
|
|
7
|
+
* prompt it was sent: the system prompt, the tool schemas, and every message up
|
|
8
|
+
* to and including itself. The only genuinely unaccounted-for text is the tail
|
|
9
|
+
* appended *after* that turn.
|
|
10
|
+
*
|
|
11
|
+
* These helpers locate the newest trustworthy usage report and tokenize only
|
|
12
|
+
* that tail, so a long session pays counting proportional to one turn instead
|
|
13
|
+
* of to the whole transcript, every turn.
|
|
14
|
+
*
|
|
15
|
+
* Trust rules for an anchor (mirroring the provider contract):
|
|
16
|
+
* - Assistant role only — nothing else carries `usage`.
|
|
17
|
+
* - Not `aborted` / `error`: those turns report partial or zero usage.
|
|
18
|
+
* - `hasContextTokenUsage(usage)`: the report must carry usable context numbers.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
|
|
22
|
+
import type { MessageCountOptions, Tokenizer } from "../tokenizer";
|
|
23
|
+
import type { AgentMessage } from "../types";
|
|
24
|
+
import { calculateContextTokens, hasContextTokenUsage } from "./compaction";
|
|
25
|
+
|
|
26
|
+
/** A provider usage report that accounts for a prefix of the transcript. */
|
|
27
|
+
export interface TranscriptUsageAnchor {
|
|
28
|
+
/** Index in the scanned array; messages at or before it are provider-accounted. */
|
|
29
|
+
index: number;
|
|
30
|
+
/** The anchoring assistant turn. */
|
|
31
|
+
message: AssistantMessage;
|
|
32
|
+
/** Conversation tokens the provider reported for that prompt. */
|
|
33
|
+
tokens: number;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Whether this message's provider usage may anchor transcript accounting.
|
|
38
|
+
*
|
|
39
|
+
* The single home for the trust rules — every anchor scan MUST route through
|
|
40
|
+
* it so a stale-usage rule can never drift between the transcript walkers and
|
|
41
|
+
* the session-entry walkers.
|
|
42
|
+
*/
|
|
43
|
+
export function isTranscriptUsageAnchor(message: AgentMessage): message is AssistantMessage {
|
|
44
|
+
if (message.role !== "assistant") return false;
|
|
45
|
+
const assistant = message as AssistantMessage;
|
|
46
|
+
if (assistant.stopReason === "aborted" || assistant.stopReason === "error") return false;
|
|
47
|
+
return assistant.usage !== undefined && hasContextTokenUsage(assistant.usage);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Newest assistant turn in `messages[fromIndex..]` whose usage can anchor the
|
|
52
|
+
* transcript, or `undefined` when none qualifies (fresh context, or every
|
|
53
|
+
* recent turn aborted/errored).
|
|
54
|
+
*
|
|
55
|
+
* `fromIndex` excludes turns whose usage is stale — anything a compaction
|
|
56
|
+
* summarized away describes a prompt that is no longer sent.
|
|
57
|
+
*/
|
|
58
|
+
export function findTranscriptUsageAnchor(
|
|
59
|
+
messages: readonly AgentMessage[],
|
|
60
|
+
fromIndex = 0,
|
|
61
|
+
): TranscriptUsageAnchor | undefined {
|
|
62
|
+
for (let index = messages.length - 1; index >= fromIndex; index--) {
|
|
63
|
+
const message = messages[index];
|
|
64
|
+
if (!isTranscriptUsageAnchor(message)) continue;
|
|
65
|
+
return { index, message, tokens: calculateContextTokens(message.usage) };
|
|
66
|
+
}
|
|
67
|
+
return undefined;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Options for {@link estimateTranscriptTokens}. */
|
|
71
|
+
export interface TranscriptTokenOptions {
|
|
72
|
+
/**
|
|
73
|
+
* Gates the anchor search only: usage at or before this index is stale (a
|
|
74
|
+
* compaction rewrote the prompt it describes) and must not anchor. Content
|
|
75
|
+
* accounting is governed separately by {@link countFromIndex}.
|
|
76
|
+
*/
|
|
77
|
+
anchorFromIndex?: number;
|
|
78
|
+
/**
|
|
79
|
+
* First message whose content is counted locally when no anchor is found.
|
|
80
|
+
* Defaults to 0 (count the whole transcript), which is what a floor
|
|
81
|
+
* estimate wants; pass the compaction boundary to skip summarized-away
|
|
82
|
+
* messages entirely.
|
|
83
|
+
*/
|
|
84
|
+
countFromIndex?: number;
|
|
85
|
+
/** Forwarded to {@link Tokenizer.countMessage} for every locally counted message. */
|
|
86
|
+
excludeEncryptedReasoning?: boolean;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Conversation tokens for `messages`: the provider's own report for everything
|
|
91
|
+
* it already covers, plus a local count of only the unaccounted-for tail.
|
|
92
|
+
*
|
|
93
|
+
* An anchored result already includes the non-message prefix (system prompt +
|
|
94
|
+
* tool schemas) because the provider charged it; an unanchored result is a
|
|
95
|
+
* message-only sum. Callers that add non-message tokens on top MUST branch on
|
|
96
|
+
* {@link findTranscriptUsageAnchor} rather than assuming one shape.
|
|
97
|
+
*/
|
|
98
|
+
export function estimateTranscriptTokens(
|
|
99
|
+
messages: readonly AgentMessage[],
|
|
100
|
+
tokenizer: Tokenizer,
|
|
101
|
+
options?: TranscriptTokenOptions,
|
|
102
|
+
): number {
|
|
103
|
+
const estimateOptions: MessageCountOptions | undefined =
|
|
104
|
+
options?.excludeEncryptedReasoning === true ? { excludeEncryptedReasoning: true } : undefined;
|
|
105
|
+
const anchor = findTranscriptUsageAnchor(messages, options?.anchorFromIndex ?? 0);
|
|
106
|
+
let total = anchor?.tokens ?? 0;
|
|
107
|
+
for (let index = anchor ? anchor.index + 1 : (options?.countFromIndex ?? 0); index < messages.length; index++) {
|
|
108
|
+
total += tokenizer.countMessage(messages[index], estimateOptions);
|
|
109
|
+
}
|
|
110
|
+
return total;
|
|
111
|
+
}
|
package/src/tokenizer.ts
CHANGED
|
@@ -1,27 +1,282 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import type { Model } from "@oh-my-pi/pi-ai";
|
|
2
|
+
import type { ModelTokenizer } from "@oh-my-pi/pi-catalog/types";
|
|
3
|
+
import { countTokens as countTokensNat, Encoding } from "@oh-my-pi/pi-natives";
|
|
4
|
+
import { stringifyJson } from "@oh-my-pi/pi-utils";
|
|
5
|
+
import * as snapcompact from "@oh-my-pi/snapcompact";
|
|
6
|
+
import { isEstimateCacheable, messageEstimateVersion } from "./compaction/message-cache";
|
|
7
|
+
import type { AgentMessage } from "./types";
|
|
2
8
|
|
|
3
|
-
const
|
|
9
|
+
const testEnv = Bun.env.NODE_ENV === "test";
|
|
10
|
+
const accurate = process.env.PI_TOKENIZER_ACCURATE === "1" && !testEnv;
|
|
4
11
|
|
|
5
|
-
|
|
12
|
+
const NATIVE_ENCODING: Record<ModelTokenizer, Encoding> = {
|
|
13
|
+
"claude-v3": Encoding.ClaudeV3,
|
|
14
|
+
"claude-v47": Encoding.ClaudeV47,
|
|
15
|
+
"claude-v5": Encoding.ClaudeV5,
|
|
16
|
+
"claude-v5-sonnet": Encoding.ClaudeV5Sonnet,
|
|
17
|
+
qwen3: Encoding.Qwen3,
|
|
18
|
+
"deepseek-v3": Encoding.DeepSeekV3,
|
|
19
|
+
"kimi-k2": Encoding.KimiK2,
|
|
20
|
+
glm5: Encoding.Glm5,
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
/** Maps the catalog-resolved tokenizer family to its native implementation. */
|
|
24
|
+
export function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): Encoding | null {
|
|
25
|
+
return model?.tokenizer ? NATIVE_ENCODING[model.tokenizer] : null;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* `strict` always pays for an exact native count (the catalog-resolved
|
|
30
|
+
* tokenizer when known, o200k_base otherwise). `approximate` and
|
|
31
|
+
* `upperbound` prefer the same exact count for known tokenizer families or
|
|
32
|
+
* when `PI_TOKENIZER_ACCURATE=1` is set; otherwise they use a cheap heuristic:
|
|
33
|
+
* `approximate` a bytes/4 guess, `upperbound` the raw byte length (never
|
|
34
|
+
* undercounts).
|
|
35
|
+
*/
|
|
36
|
+
export type TokenCountMode = "strict" | "approximate" | "upperbound";
|
|
37
|
+
|
|
38
|
+
/** Options for {@link Tokenizer.countMessage} / {@link Tokenizer.countMessages}. */
|
|
39
|
+
export interface MessageCountOptions {
|
|
40
|
+
/**
|
|
41
|
+
* Drop opaque provider reasoning payloads (`thinkingSignature`,
|
|
42
|
+
* `redactedThinking`, native server-tool blocks) from the estimate. Those
|
|
43
|
+
* are billed by the provider on replay, so the default counts them — but
|
|
44
|
+
* their *local* byte size can diverge wildly from what the provider
|
|
45
|
+
* charges, so the compaction floor (which only needs the reliably-countable,
|
|
46
|
+
* on-wire-compressible content) excludes them to avoid false triggers on
|
|
47
|
+
* thinking-heavy turns.
|
|
48
|
+
*/
|
|
49
|
+
excludeEncryptedReasoning?: boolean;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
function byteEstimate(text: string): number {
|
|
6
53
|
return (Buffer.byteLength(text, "utf-8") + 3) >> 2;
|
|
7
54
|
}
|
|
8
55
|
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
return countTokensNat(text);
|
|
12
|
-
} else if (Array.isArray(text)) {
|
|
13
|
-
return text.reduce((sum, t) => sum + estimateTokens(t), 0);
|
|
14
|
-
} else {
|
|
15
|
-
return estimateTokens(text);
|
|
16
|
-
}
|
|
56
|
+
function byteLength(text: string): number {
|
|
57
|
+
return Buffer.byteLength(text, "utf-8");
|
|
17
58
|
}
|
|
18
59
|
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
60
|
+
function sumFragments(text: string | string[], perFragment: (t: string) => number): number {
|
|
61
|
+
return Array.isArray(text) ? text.reduce((sum, t) => sum + perFragment(t), 0) : perFragment(text);
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/** Verdict from {@link Tokenizer.checkTokenBudget}. */
|
|
65
|
+
export interface TokenBudgetCheck {
|
|
66
|
+
/** Whether the text fits the budget. */
|
|
67
|
+
fits: boolean;
|
|
68
|
+
/**
|
|
69
|
+
* Token count behind the verdict: the exact native count when `exact` is
|
|
70
|
+
* set, otherwise the cheap byte upper bound (which already fit, so it is
|
|
71
|
+
* only an over-estimate of a count known to be under budget).
|
|
72
|
+
*/
|
|
73
|
+
tokens: number;
|
|
74
|
+
/** Whether the exact tokenizer had to run because the cheap bound busted. */
|
|
75
|
+
exact: boolean;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Image content has no tokenizer representation; charge a fixed estimate
|
|
80
|
+
* matching what providers typically bill for inline images.
|
|
81
|
+
*/
|
|
82
|
+
const IMAGE_TOKEN_ESTIMATE = 1200;
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Memoized estimates for one message under this tokenizer's encoding, split by
|
|
86
|
+
* the {@link MessageCountOptions.excludeEncryptedReasoning} option so the two
|
|
87
|
+
* variants never collide. `version` snapshots {@link messageEstimateVersion} at
|
|
88
|
+
* write time; an owner mutation (prune/shake/strip-images) bumps the version,
|
|
89
|
+
* which invalidates the entry in every live Tokenizer at once.
|
|
90
|
+
*/
|
|
91
|
+
interface MessageEstimate {
|
|
92
|
+
version: number;
|
|
93
|
+
default?: number;
|
|
94
|
+
floored?: number;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* Model-aware local token counter. Immutable: the catalog-resolved encoding
|
|
99
|
+
* is fixed at construction, so a cached count can never straddle two
|
|
100
|
+
* encodings. An `Agent` owns one for its active model (swapping the instance
|
|
101
|
+
* when the model's encoding changes); one-shot flows construct their own for
|
|
102
|
+
* the model that will be billed. Known tokenizer families use exact native
|
|
103
|
+
* counts; unknown models keep the fast byte estimate (or o200k when
|
|
104
|
+
* `PI_TOKENIZER_ACCURATE=1`).
|
|
105
|
+
*/
|
|
106
|
+
export class Tokenizer {
|
|
107
|
+
readonly #encoding: Encoding | null;
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Per-message estimate memo. Keyed by message identity, deliberately not a
|
|
111
|
+
* symbol-tagged property: callers spread messages to derive throwaway
|
|
112
|
+
* variants for counting (`estimateBranchSummaryTokens` does
|
|
113
|
+
* `countMessage({ ...message, content: truncated })`), and a property-borne
|
|
114
|
+
* cache would ride along the spread. Identity keying gives clones a fresh
|
|
115
|
+
* count.
|
|
116
|
+
*/
|
|
117
|
+
#estimates = new WeakMap<AgentMessage, MessageEstimate>();
|
|
118
|
+
|
|
119
|
+
constructor(model?: Pick<Model, "tokenizer"> | null) {
|
|
120
|
+
this.#encoding = tokenizerEncodingForModel(model);
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
get encoding(): Encoding | null {
|
|
124
|
+
return this.#encoding;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
countTokens(text: string | string[], mode: TokenCountMode = "approximate"): number {
|
|
128
|
+
if (mode === "strict") return countTokensNat(text, this.#encoding);
|
|
129
|
+
if (!testEnv && this.#encoding !== null) return countTokensNat(text, this.#encoding);
|
|
130
|
+
if (accurate) return countTokensNat(text);
|
|
131
|
+
return sumFragments(text, mode === "upperbound" ? byteLength : byteEstimate);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Cheap-first budget probe — the way to ask "does this fit in `budget`
|
|
136
|
+
* tokens?" without tokenizing the world.
|
|
137
|
+
*
|
|
138
|
+
* Byte length is a hard upper bound on token count (every token consumes at
|
|
139
|
+
* least one input byte), so text whose raw bytes already fit the budget
|
|
140
|
+
* cannot possibly exceed it — that verdict is returned without tokenizing at
|
|
141
|
+
* all. Only text that busts the bound is ambiguous, and only that case pays
|
|
142
|
+
* for the exact count. Since the bound overshoots ~4x on ordinary prose, the
|
|
143
|
+
* common "comfortably under budget" answer is free.
|
|
144
|
+
*/
|
|
145
|
+
checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck {
|
|
146
|
+
const bound = sumFragments(text, byteLength);
|
|
147
|
+
if (bound <= budget) return { fits: true, tokens: bound, exact: false };
|
|
148
|
+
const tokens = this.countTokens(text, "strict");
|
|
149
|
+
return { fits: tokens <= budget, tokens, exact: true };
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* Token estimate for one message under this tokenizer's encoding.
|
|
154
|
+
*
|
|
155
|
+
* Settled historical messages are counted once and reused until an owner
|
|
156
|
+
* (prune/shake/strip-images) calls `invalidateMessageCache`; streaming
|
|
157
|
+
* assistants bypass the memo entirely (see the message-cache settle-gate
|
|
158
|
+
* invariant). Image blocks charge a fixed per-image estimate.
|
|
159
|
+
*/
|
|
160
|
+
countMessage(message: AgentMessage, options?: MessageCountOptions): number {
|
|
161
|
+
const floored = options?.excludeEncryptedReasoning === true;
|
|
162
|
+
if (!isEstimateCacheable(message)) return this.#measureMessage(message, floored);
|
|
163
|
+
const version = messageEstimateVersion(message);
|
|
164
|
+
let entry = this.#estimates.get(message);
|
|
165
|
+
if (entry === undefined || entry.version !== version) {
|
|
166
|
+
entry = { version };
|
|
167
|
+
this.#estimates.set(message, entry);
|
|
168
|
+
}
|
|
169
|
+
const cached = floored ? entry.floored : entry.default;
|
|
170
|
+
if (cached !== undefined) return cached;
|
|
171
|
+
const result = this.#measureMessage(message, floored);
|
|
172
|
+
if (floored) entry.floored = result;
|
|
173
|
+
else entry.default = result;
|
|
174
|
+
return result;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/** Sum of {@link countMessage} over `messages`. */
|
|
178
|
+
countMessages(messages: readonly AgentMessage[], options?: MessageCountOptions): number {
|
|
179
|
+
let total = 0;
|
|
180
|
+
for (const message of messages) total += this.countMessage(message, options);
|
|
181
|
+
return total;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
#measureMessage(message: AgentMessage, excludeEncryptedReasoning: boolean): number {
|
|
185
|
+
const fragments: string[] = [];
|
|
186
|
+
let extra = 0;
|
|
187
|
+
// Declaration-merged app roles (the coding-agent's bashExecution) are
|
|
188
|
+
// invisible to this package's union, so the discriminant is read as data.
|
|
189
|
+
const role: string = message.role;
|
|
190
|
+
if (role === "bashExecution") {
|
|
191
|
+
if ("command" in message && typeof message.command === "string") fragments.push(message.command);
|
|
192
|
+
if ("output" in message && typeof message.output === "string") fragments.push(message.output);
|
|
193
|
+
return fragments.length === 0 ? 0 : this.countTokens(fragments);
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
switch (message.role) {
|
|
197
|
+
case "user": {
|
|
198
|
+
const content: string | Array<{ type: string; text?: string }> = message.content;
|
|
199
|
+
if (typeof content === "string") {
|
|
200
|
+
fragments.push(content);
|
|
201
|
+
} else if (Array.isArray(content)) {
|
|
202
|
+
for (const block of content) {
|
|
203
|
+
if (block.type === "text" && block.text) {
|
|
204
|
+
fragments.push(block.text);
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
break;
|
|
209
|
+
}
|
|
210
|
+
case "assistant": {
|
|
211
|
+
for (const block of message.content) {
|
|
212
|
+
if (block.type === "text") {
|
|
213
|
+
fragments.push(block.text);
|
|
214
|
+
} else if (block.type === "thinking") {
|
|
215
|
+
fragments.push(block.thinking);
|
|
216
|
+
// Providers charge for the opaque signature/reasoning payload that
|
|
217
|
+
// rides alongside the thinking text (OpenAI Responses encrypted
|
|
218
|
+
// reasoning items, Anthropic signed thinking blocks, etc.). Without
|
|
219
|
+
// counting it, this estimator can read ~half of the provider-reported
|
|
220
|
+
// usage on thinking-heavy turns — see #2275 for the resulting
|
|
221
|
+
// compaction-trigger / post-check metric divergence. The compaction
|
|
222
|
+
// floor excludes it (its local byte size diverges from provider billing).
|
|
223
|
+
if (block.thinkingSignature && !excludeEncryptedReasoning) {
|
|
224
|
+
fragments.push(block.thinkingSignature);
|
|
225
|
+
}
|
|
226
|
+
} else if (block.type === "toolCall") {
|
|
227
|
+
fragments.push(block.name);
|
|
228
|
+
fragments.push(stringifyJson(block.arguments) ?? "null");
|
|
229
|
+
} else if (block.type === "redactedThinking") {
|
|
230
|
+
// Encrypted reasoning blob the provider still bills for on replay;
|
|
231
|
+
// excluded from the compaction floor for the same reason as above.
|
|
232
|
+
if (!excludeEncryptedReasoning) fragments.push(block.data);
|
|
233
|
+
} else if (block.type === "anthropicServerTool") {
|
|
234
|
+
// Native Anthropic server-tool call/result replayed verbatim on the
|
|
235
|
+
// wire (server_tool_use input and opaque result content). The provider
|
|
236
|
+
// still bills for it on same-provider replay; excluded from the
|
|
237
|
+
// compaction floor like other encrypted reasoning because its local
|
|
238
|
+
// byte size diverges from provider billing.
|
|
239
|
+
if (!excludeEncryptedReasoning) fragments.push(stringifyJson(block.block) ?? "null");
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
break;
|
|
243
|
+
}
|
|
244
|
+
case "hookMessage":
|
|
245
|
+
case "toolResult": {
|
|
246
|
+
if (typeof message.content === "string") {
|
|
247
|
+
fragments.push(message.content);
|
|
248
|
+
} else {
|
|
249
|
+
for (const block of message.content) {
|
|
250
|
+
if (block.type === "text" && block.text) {
|
|
251
|
+
fragments.push(block.text);
|
|
252
|
+
} else if (block.type === "image") {
|
|
253
|
+
extra += IMAGE_TOKEN_ESTIMATE;
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
break;
|
|
258
|
+
}
|
|
259
|
+
case "branchSummary":
|
|
260
|
+
case "compactionSummary": {
|
|
261
|
+
fragments.push(message.summary);
|
|
262
|
+
if (message.role === "compactionSummary") {
|
|
263
|
+
if (message.blocks) {
|
|
264
|
+
for (const block of message.blocks) {
|
|
265
|
+
if (block.type === "text") fragments.push(block.text);
|
|
266
|
+
else extra += snapcompact.FRAME_TOKEN_ESTIMATE;
|
|
267
|
+
}
|
|
268
|
+
} else if (message.images) {
|
|
269
|
+
// Snapcompact frames render at ≥1568px; providers bill the downscaled cap.
|
|
270
|
+
extra += message.images.length * snapcompact.FRAME_TOKEN_ESTIMATE;
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
break;
|
|
274
|
+
}
|
|
275
|
+
default:
|
|
276
|
+
return 0;
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
if (fragments.length === 0) return extra;
|
|
280
|
+
return extra + this.countTokens(fragments);
|
|
26
281
|
}
|
|
27
282
|
}
|