@oh-my-pi/pi-agent-core 17.3.8 → 17.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/dist/types/agent.d.ts +8 -1
- package/dist/types/compaction/branch-summarization.d.ts +2 -1
- package/dist/types/compaction/compaction.d.ts +24 -17
- package/dist/types/compaction/index.d.ts +1 -0
- package/dist/types/compaction/message-cache.d.ts +6 -4
- package/dist/types/compaction/messages.d.ts +17 -1
- package/dist/types/compaction/openai.d.ts +2 -1
- package/dist/types/compaction/pruning.d.ts +3 -2
- package/dist/types/compaction/shake.d.ts +2 -1
- package/dist/types/compaction/transcript-tokens.d.ts +76 -0
- package/dist/types/tokenizer.d.ts +78 -2
- package/dist/types/types.d.ts +1 -1
- package/package.json +8 -8
- package/src/agent.ts +25 -4
- package/src/compaction/branch-summarization.ts +15 -8
- package/src/compaction/compaction.ts +43 -152
- package/src/compaction/index.ts +1 -0
- package/src/compaction/message-cache.ts +33 -34
- package/src/compaction/messages.ts +21 -5
- package/src/compaction/openai.ts +46 -18
- package/src/compaction/pruning.ts +25 -13
- package/src/compaction/shake.ts +18 -16
- package/src/compaction/transcript-tokens.ts +111 -0
- package/src/tokenizer.ts +273 -18
- package/src/types.ts +1 -1
|
@@ -31,11 +31,11 @@ import { buildResponsesInput, resolveOpenAICompatPolicy } from "@oh-my-pi/pi-ai/
|
|
|
31
31
|
import { stripOpenAIResponsesOutputOnlyStatusesForReplay } from "@oh-my-pi/pi-ai/utils";
|
|
32
32
|
import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
|
|
33
33
|
import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking";
|
|
34
|
-
import { isRecord, logger, prompt
|
|
34
|
+
import { isRecord, logger, prompt } from "@oh-my-pi/pi-utils";
|
|
35
35
|
import * as snapcompact from "@oh-my-pi/snapcompact";
|
|
36
36
|
import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
|
|
37
37
|
import { ThinkingLevel } from "../thinking";
|
|
38
|
-
import {
|
|
38
|
+
import { Tokenizer } from "../tokenizer";
|
|
39
39
|
import type { AgentMessage } from "../types";
|
|
40
40
|
import {
|
|
41
41
|
buildCompactionV2Request,
|
|
@@ -47,7 +47,6 @@ import {
|
|
|
47
47
|
} from "./compaction-v2-streaming";
|
|
48
48
|
import type { CompactionEntry, SessionEntry } from "./entries";
|
|
49
49
|
import { NativeCompactionError } from "./errors";
|
|
50
|
-
import { isEstimateCacheable, readEstimateCache, writeEstimateCache } from "./message-cache";
|
|
51
50
|
import { type ConvertToLlm, createBranchSummaryMessage, createCustomMessage, defaultConvertToLlm } from "./messages";
|
|
52
51
|
import {
|
|
53
52
|
buildOpenAiNativeHistory,
|
|
@@ -390,144 +389,17 @@ export function resolveThresholdTokens(contextWindow: number, settings: Compacti
|
|
|
390
389
|
// Cut point detection
|
|
391
390
|
// ============================================================================
|
|
392
391
|
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
/**
|
|
400
|
-
* Estimate token count for a message using cl100k_base via the native
|
|
401
|
-
* tokenizer. This is not Claude's first-party tokenizer (Anthropic doesn't
|
|
402
|
-
* publish one) but is within ~5–10% across English/code text.
|
|
403
|
-
*
|
|
404
|
-
* `excludeEncryptedReasoning` drops opaque provider reasoning payloads
|
|
405
|
-
* (`thinkingSignature`, `redactedThinking`) from the estimate. Those are billed
|
|
406
|
-
* by the provider on replay, so the default counts them — but their *local*
|
|
407
|
-
* byte size can diverge wildly from what the provider charges, so the
|
|
408
|
-
* compaction floor (which only needs the reliably-countable, on-wire-compressible
|
|
409
|
-
* content) excludes them to avoid false triggers on thinking-heavy turns.
|
|
410
|
-
*/
|
|
411
|
-
export function estimateTokens(message: AgentMessage, options?: { excludeEncryptedReasoning?: boolean }): number {
|
|
412
|
-
// Settled historical messages are counted once and reused until an owner
|
|
413
|
-
// (prune/shake/strip-images) invalidates them; streaming assistants bypass
|
|
414
|
-
// the cache entirely (see message-cache.ts settle-gate invariant).
|
|
415
|
-
const cacheable = isEstimateCacheable(message);
|
|
416
|
-
const excludeEncryptedReasoning = options?.excludeEncryptedReasoning === true;
|
|
417
|
-
if (cacheable) {
|
|
418
|
-
const cached = readEstimateCache(message, excludeEncryptedReasoning);
|
|
419
|
-
if (cached !== undefined) return cached;
|
|
420
|
-
}
|
|
421
|
-
const result = computeMessageTokens(message, options);
|
|
422
|
-
if (cacheable) writeEstimateCache(message, excludeEncryptedReasoning, result);
|
|
423
|
-
return result;
|
|
424
|
-
}
|
|
425
|
-
|
|
426
|
-
function computeMessageTokens(message: AgentMessage, options?: { excludeEncryptedReasoning?: boolean }): number {
|
|
427
|
-
const fragments: string[] = [];
|
|
428
|
-
let extra = 0;
|
|
429
|
-
if ((message as { role?: string }).role === "bashExecution") {
|
|
430
|
-
const bash = message as { command?: unknown; output?: unknown };
|
|
431
|
-
if (typeof bash.command === "string") fragments.push(bash.command);
|
|
432
|
-
if (typeof bash.output === "string") fragments.push(bash.output);
|
|
433
|
-
return fragments.length === 0 ? 0 : countTokens(fragments);
|
|
434
|
-
}
|
|
435
|
-
|
|
436
|
-
switch (message.role) {
|
|
437
|
-
case "user": {
|
|
438
|
-
const content = (message as { content: string | Array<{ type: string; text?: string }> }).content;
|
|
439
|
-
if (typeof content === "string") {
|
|
440
|
-
fragments.push(content);
|
|
441
|
-
} else if (Array.isArray(content)) {
|
|
442
|
-
for (const block of content) {
|
|
443
|
-
if (block.type === "text" && block.text) {
|
|
444
|
-
fragments.push(block.text);
|
|
445
|
-
}
|
|
446
|
-
}
|
|
447
|
-
}
|
|
448
|
-
break;
|
|
449
|
-
}
|
|
450
|
-
case "assistant": {
|
|
451
|
-
const assistant = message as AssistantMessage;
|
|
452
|
-
for (const block of assistant.content) {
|
|
453
|
-
if (block.type === "text") {
|
|
454
|
-
fragments.push(block.text);
|
|
455
|
-
} else if (block.type === "thinking") {
|
|
456
|
-
fragments.push(block.thinking);
|
|
457
|
-
// Providers charge for the opaque signature/reasoning payload that
|
|
458
|
-
// rides alongside the thinking text (OpenAI Responses encrypted
|
|
459
|
-
// reasoning items, Anthropic signed thinking blocks, etc.). Without
|
|
460
|
-
// counting it, this estimator can read ~half of the provider-reported
|
|
461
|
-
// usage on thinking-heavy turns — see #2275 for the resulting
|
|
462
|
-
// compaction-trigger / post-check metric divergence. The compaction
|
|
463
|
-
// floor excludes it (its local byte size diverges from provider billing).
|
|
464
|
-
if (block.thinkingSignature && !options?.excludeEncryptedReasoning) {
|
|
465
|
-
fragments.push(block.thinkingSignature);
|
|
466
|
-
}
|
|
467
|
-
} else if (block.type === "toolCall") {
|
|
468
|
-
fragments.push(block.name);
|
|
469
|
-
fragments.push(stringifyJson(block.arguments) ?? "null");
|
|
470
|
-
} else if (block.type === "redactedThinking") {
|
|
471
|
-
// Encrypted reasoning blob the provider still bills for on replay;
|
|
472
|
-
// excluded from the compaction floor for the same reason as above.
|
|
473
|
-
if (!options?.excludeEncryptedReasoning) fragments.push(block.data);
|
|
474
|
-
} else if (block.type === "anthropicServerTool") {
|
|
475
|
-
// Native Anthropic server-tool call/result replayed verbatim on the
|
|
476
|
-
// wire (server_tool_use input and opaque result content). This opaque
|
|
477
|
-
// provider-replay state the provider still
|
|
478
|
-
// bills for on same-provider replay; excluded from the compaction
|
|
479
|
-
// floor like other encrypted reasoning because its local byte size
|
|
480
|
-
// diverges from provider billing.
|
|
481
|
-
if (!options?.excludeEncryptedReasoning) fragments.push(stringifyJson(block.block) ?? "null");
|
|
482
|
-
}
|
|
483
|
-
}
|
|
484
|
-
break;
|
|
485
|
-
}
|
|
486
|
-
case "hookMessage":
|
|
487
|
-
case "toolResult": {
|
|
488
|
-
if (typeof message.content === "string") {
|
|
489
|
-
fragments.push(message.content);
|
|
490
|
-
} else {
|
|
491
|
-
for (const block of message.content) {
|
|
492
|
-
if (block.type === "text" && block.text) {
|
|
493
|
-
fragments.push(block.text);
|
|
494
|
-
} else if (block.type === "image") {
|
|
495
|
-
extra += IMAGE_TOKEN_ESTIMATE;
|
|
496
|
-
}
|
|
497
|
-
}
|
|
498
|
-
}
|
|
499
|
-
break;
|
|
500
|
-
}
|
|
501
|
-
case "branchSummary":
|
|
502
|
-
case "compactionSummary": {
|
|
503
|
-
fragments.push(message.summary);
|
|
504
|
-
if (message.role === "compactionSummary") {
|
|
505
|
-
if (message.blocks) {
|
|
506
|
-
for (const block of message.blocks) {
|
|
507
|
-
if (block.type === "text") fragments.push(block.text);
|
|
508
|
-
else extra += snapcompact.FRAME_TOKEN_ESTIMATE;
|
|
509
|
-
}
|
|
510
|
-
} else if (message.images) {
|
|
511
|
-
// Snapcompact frames render at ≥1568px; providers bill the downscaled cap.
|
|
512
|
-
extra += message.images.length * snapcompact.FRAME_TOKEN_ESTIMATE;
|
|
513
|
-
}
|
|
514
|
-
}
|
|
515
|
-
break;
|
|
516
|
-
}
|
|
517
|
-
default:
|
|
518
|
-
return 0;
|
|
519
|
-
}
|
|
520
|
-
|
|
521
|
-
if (fragments.length === 0) return extra;
|
|
522
|
-
return extra + countTokens(fragments);
|
|
523
|
-
}
|
|
524
|
-
|
|
525
|
-
function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number {
|
|
392
|
+
function estimateEntriesTokens(
|
|
393
|
+
entries: SessionEntry[],
|
|
394
|
+
tokenizer: Tokenizer,
|
|
395
|
+
startIndex: number,
|
|
396
|
+
endIndex: number,
|
|
397
|
+
): number {
|
|
526
398
|
let total = 0;
|
|
527
399
|
for (let i = startIndex; i < endIndex; i++) {
|
|
528
400
|
const msg = getMessageFromEntry(entries[i]);
|
|
529
401
|
if (msg) {
|
|
530
|
-
total +=
|
|
402
|
+
total += tokenizer.countMessage(msg);
|
|
531
403
|
}
|
|
532
404
|
}
|
|
533
405
|
return total;
|
|
@@ -626,6 +498,7 @@ export interface CutPointResult {
|
|
|
626
498
|
*/
|
|
627
499
|
export function findCutPoint(
|
|
628
500
|
entries: SessionEntry[],
|
|
501
|
+
tokenizer: Tokenizer,
|
|
629
502
|
startIndex: number,
|
|
630
503
|
endIndex: number,
|
|
631
504
|
keepRecentTokens: number,
|
|
@@ -645,7 +518,7 @@ export function findCutPoint(
|
|
|
645
518
|
if (entry.type !== "message") continue;
|
|
646
519
|
|
|
647
520
|
// Estimate this message's size
|
|
648
|
-
const messageTokens =
|
|
521
|
+
const messageTokens = tokenizer.countMessage(entry.message);
|
|
649
522
|
accumulatedTokens += messageTokens;
|
|
650
523
|
|
|
651
524
|
// Check if we've exceeded the budget
|
|
@@ -948,12 +821,17 @@ interface SummaryWindow {
|
|
|
948
821
|
* on message boundaries. Only called when the whole conversation does not fit —
|
|
949
822
|
* the common single-window path never pays this per-message sizing pass.
|
|
950
823
|
*/
|
|
951
|
-
function planSummaryWindows(
|
|
824
|
+
function planSummaryWindows(
|
|
825
|
+
messages: Message[],
|
|
826
|
+
tokenizer: Tokenizer,
|
|
827
|
+
dialect: Dialect | undefined,
|
|
828
|
+
budgetTokens: number,
|
|
829
|
+
): Message[][] {
|
|
952
830
|
const windows: Message[][] = [];
|
|
953
831
|
let current: Message[] = [];
|
|
954
832
|
let currentTokens = 0;
|
|
955
833
|
for (const message of messages) {
|
|
956
|
-
const tokens = countTokens(serializeConversationForSummary([message], dialect));
|
|
834
|
+
const tokens = tokenizer.countTokens(serializeConversationForSummary([message], dialect));
|
|
957
835
|
if (currentTokens > 0 && currentTokens + tokens > budgetTokens) {
|
|
958
836
|
windows.push(current);
|
|
959
837
|
current = [];
|
|
@@ -982,6 +860,7 @@ export async function generateSummary(
|
|
|
982
860
|
// Convert to LLM messages first (handles custom app messages when caller provides a transformer).
|
|
983
861
|
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(currentMessages);
|
|
984
862
|
const dialect = preferredDialect(model.id);
|
|
863
|
+
const tokenizer = new Tokenizer(model);
|
|
985
864
|
const wholeConversation = serializeConversationForSummary(llmMessages, dialect);
|
|
986
865
|
const budgetTokens = summaryInputBudgetTokens(model, maxTokens);
|
|
987
866
|
// A span that outgrew the summarizer's window is summarized as a fold: each
|
|
@@ -991,19 +870,21 @@ export async function generateSummary(
|
|
|
991
870
|
// retry can shrink — the state a cross-provider compaction boundary
|
|
992
871
|
// (see `prepareCompaction`) puts a long session into. One window is the
|
|
993
872
|
// common case and costs exactly the one call it always did.
|
|
994
|
-
const pending: SummaryWindow[] =
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
: planSummaryWindows(llmMessages, dialect, budgetTokens).map(messages => ({ messages, budgetTokens }));
|
|
873
|
+
const pending: SummaryWindow[] = tokenizer.checkTokenBudget(wholeConversation, budgetTokens).fits
|
|
874
|
+
? [{ messages: llmMessages, budgetTokens, text: wholeConversation }]
|
|
875
|
+
: planSummaryWindows(llmMessages, tokenizer, dialect, budgetTokens).map(messages => ({ messages, budgetTokens }));
|
|
998
876
|
|
|
999
877
|
let carriedSummary = previousSummary;
|
|
1000
878
|
while (pending.length > 0) {
|
|
1001
879
|
const window = pending[0];
|
|
1002
880
|
const text = window.text ?? serializeConversationForSummary(window.messages, dialect);
|
|
1003
|
-
|
|
881
|
+
// A budget probe, not a raw count: a window whose bytes already fit needs
|
|
882
|
+
// neither an exact count nor the clamp, and the bust path hands back the
|
|
883
|
+
// exact count the proportional clamp needs as its denominator.
|
|
884
|
+
const budget = tokenizer.checkTokenBudget(text, window.budgetTokens);
|
|
1004
885
|
try {
|
|
1005
886
|
carriedSummary = await summarizeConversationWindow(
|
|
1006
|
-
clampConversationToBudget(text, window.budgetTokens,
|
|
887
|
+
budget.fits ? text : clampConversationToBudget(text, window.budgetTokens, budget.tokens),
|
|
1007
888
|
carriedSummary,
|
|
1008
889
|
model,
|
|
1009
890
|
maxTokens,
|
|
@@ -1020,8 +901,11 @@ export async function generateSummary(
|
|
|
1020
901
|
// window size only the provider can tell us is wrong.
|
|
1021
902
|
// Halve what was actually SENT, not the budget it was planned against:
|
|
1022
903
|
// the rejection proves the plan was fiction, so converging on the real
|
|
1023
|
-
// cap must not spend a call per level of an imaginary ladder.
|
|
1024
|
-
|
|
904
|
+
// cap must not spend a call per level of an imaginary ladder. The cheap
|
|
905
|
+
// fit path never counted this window, so pay for the exact size here —
|
|
906
|
+
// one tokenization is nothing against the provider round trip already lost.
|
|
907
|
+
const sentTokens = budget.exact ? budget.tokens : tokenizer.countTokens(text, "strict");
|
|
908
|
+
const halved = Math.floor(Math.min(window.budgetTokens, sentTokens) / 2);
|
|
1025
909
|
if (
|
|
1026
910
|
!AIError.is(AIError.classify(error), AIError.Flag.ContextOverflow) ||
|
|
1027
911
|
halved < minSummaryInputTokens(model)
|
|
@@ -1031,7 +915,7 @@ export async function generateSummary(
|
|
|
1031
915
|
pending.splice(
|
|
1032
916
|
0,
|
|
1033
917
|
1,
|
|
1034
|
-
...planSummaryWindows(window.messages, dialect, halved).map(messages => ({
|
|
918
|
+
...planSummaryWindows(window.messages, tokenizer, dialect, halved).map(messages => ({
|
|
1035
919
|
messages,
|
|
1036
920
|
budgetTokens: halved,
|
|
1037
921
|
})),
|
|
@@ -1384,7 +1268,7 @@ export interface CompactionPreparation {
|
|
|
1384
1268
|
* let the active model replay it, so keying reuse on "any candidate shares the
|
|
1385
1269
|
* provider" left a provider-switched session permanently context-less (#6343).
|
|
1386
1270
|
*/
|
|
1387
|
-
function remotePreserveReusable(
|
|
1271
|
+
export function remotePreserveReusable(
|
|
1388
1272
|
preserveData: Record<string, unknown> | undefined,
|
|
1389
1273
|
activeModel: Model,
|
|
1390
1274
|
settings: CompactionSettings,
|
|
@@ -1423,10 +1307,16 @@ export function findReadableCompactionIndex(
|
|
|
1423
1307
|
return -1;
|
|
1424
1308
|
}
|
|
1425
1309
|
|
|
1310
|
+
/**
|
|
1311
|
+
* Pass the caller's warm `tokenizer` (the Agent's for the active model) so the
|
|
1312
|
+
* full-branch estimate walk hits its memo; the cold default is for one-shot
|
|
1313
|
+
* callers that have no live agent.
|
|
1314
|
+
*/
|
|
1426
1315
|
export function prepareCompaction(
|
|
1427
1316
|
pathEntries: SessionEntry[],
|
|
1428
1317
|
settings: CompactionSettings,
|
|
1429
1318
|
activeModel?: Model,
|
|
1319
|
+
tokenizer: Tokenizer = new Tokenizer(activeModel),
|
|
1430
1320
|
): CompactionPreparation | undefined {
|
|
1431
1321
|
if (pathEntries.length > 0 && pathEntries[pathEntries.length - 1].type === "compaction") {
|
|
1432
1322
|
return undefined;
|
|
@@ -1459,7 +1349,7 @@ export function prepareCompaction(
|
|
|
1459
1349
|
const tokensBefore = lastUsage ? calculateContextTokens(lastUsage) : 0;
|
|
1460
1350
|
let keepRecentTokens = settings.keepRecentTokens;
|
|
1461
1351
|
if (lastUsage) {
|
|
1462
|
-
const estimatedTokens = estimateEntriesTokens(pathEntries, boundaryStart, boundaryEnd);
|
|
1352
|
+
const estimatedTokens = estimateEntriesTokens(pathEntries, tokenizer, boundaryStart, boundaryEnd);
|
|
1463
1353
|
const promptTokens = calculatePromptTokens(lastUsage);
|
|
1464
1354
|
const ratio = estimatedTokens > 0 ? promptTokens / estimatedTokens : 0;
|
|
1465
1355
|
if (Number.isFinite(ratio) && ratio > 1) {
|
|
@@ -1467,7 +1357,7 @@ export function prepareCompaction(
|
|
|
1467
1357
|
}
|
|
1468
1358
|
}
|
|
1469
1359
|
|
|
1470
|
-
const cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, keepRecentTokens);
|
|
1360
|
+
const cutPoint = findCutPoint(pathEntries, tokenizer, boundaryStart, boundaryEnd, keepRecentTokens);
|
|
1471
1361
|
|
|
1472
1362
|
// Get ID of first kept entry
|
|
1473
1363
|
const firstKeptEntry = pathEntries[cutPoint.firstKeptEntryIndex];
|
|
@@ -1706,6 +1596,7 @@ export async function compact(
|
|
|
1706
1596
|
: undefined;
|
|
1707
1597
|
const trimmed = trimRemoteCompactionInputToContextWindow(
|
|
1708
1598
|
remoteHistory,
|
|
1599
|
+
new Tokenizer(model),
|
|
1709
1600
|
model.contextWindow,
|
|
1710
1601
|
instructions,
|
|
1711
1602
|
tools,
|
package/src/compaction/index.ts
CHANGED
|
@@ -1,13 +1,12 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
3
|
-
* ({@link
|
|
2
|
+
* Cache-coherence seams for the two hot history walks: token estimation
|
|
3
|
+
* ({@link Tokenizer.countMessage}) and LLM conversion (the coding-agent's
|
|
4
|
+
* `convertToLlm`).
|
|
4
5
|
*
|
|
5
6
|
* Long sessions re-walk a settled `AgentMessage[]` every turn, re-tokenizing and
|
|
6
|
-
* re-converting historical objects that only the newest suffix can change.
|
|
7
|
-
*
|
|
8
|
-
* and
|
|
9
|
-
*
|
|
10
|
-
* Correctness rests on two invariants:
|
|
7
|
+
* re-converting historical objects that only the newest suffix can change. Each
|
|
8
|
+
* `Tokenizer` memoizes estimates per message identity; this module owns the two
|
|
9
|
+
* invariants that keep those memos (and the cross-package convert memo) honest:
|
|
11
10
|
*
|
|
12
11
|
* 1. **Settle gate.** A streaming assistant is mutated under one identity while
|
|
13
12
|
* its `usage`/`stopReason` are provisional (the seed carries zeroed usage and
|
|
@@ -19,9 +18,11 @@
|
|
|
19
18
|
* 2. **Owner invalidation.** `pruneToolOutputs` / `pruneSupersededToolResults`,
|
|
20
19
|
* `applyShakeRegion`, and `stripImagesFromMessage` rewrite message content in
|
|
21
20
|
* place under a stable identity. Each MUST call {@link invalidateMessageCache}
|
|
22
|
-
* on the mutated message before the next convert/estimate pass
|
|
23
|
-
*
|
|
24
|
-
*
|
|
21
|
+
* on the mutated message before the next convert/estimate pass. Invalidation
|
|
22
|
+
* bumps a symbol-keyed version tag on the message itself, so every live
|
|
23
|
+
* `Tokenizer` memo drops its stale entry at once without registering
|
|
24
|
+
* anywhere; the convert cache lives in another package and subscribes via
|
|
25
|
+
* {@link registerMessageCacheInvalidator}.
|
|
25
26
|
*/
|
|
26
27
|
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
|
|
27
28
|
import type { AgentMessage } from "../types";
|
|
@@ -41,18 +42,26 @@ export function registerMessageCacheInvalidator(invalidate: (message: AgentMessa
|
|
|
41
42
|
};
|
|
42
43
|
}
|
|
43
44
|
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
45
|
+
/**
|
|
46
|
+
* Estimate-version tag riding on the message itself. Symbol-keyed, so JSON
|
|
47
|
+
* session persistence and default iteration never see it. Object spread copies
|
|
48
|
+
* the tag onto derived clones — harmless, because estimate memos key on message
|
|
49
|
+
* *identity* and a fresh clone starts with no memo entries anywhere.
|
|
50
|
+
*/
|
|
51
|
+
const kEstimateVersion = Symbol("omp.messageEstimateVersion");
|
|
52
|
+
|
|
53
|
+
interface VersionedMessage {
|
|
54
|
+
[kEstimateVersion]?: number;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Current estimate version of `message` (0 until first invalidation). A
|
|
59
|
+
* `Tokenizer` memo entry stamped with an older version is stale and must be
|
|
60
|
+
* recounted.
|
|
61
|
+
*/
|
|
62
|
+
export function messageEstimateVersion(message: AgentMessage): number {
|
|
63
|
+
return (message as VersionedMessage)[kEstimateVersion] ?? 0;
|
|
64
|
+
}
|
|
56
65
|
|
|
57
66
|
/**
|
|
58
67
|
* True when this message's estimate is safe to cache by identity. Non-assistants
|
|
@@ -70,23 +79,13 @@ export function isEstimateCacheable(message: AgentMessage): boolean {
|
|
|
70
79
|
);
|
|
71
80
|
}
|
|
72
81
|
|
|
73
|
-
/** Read a cached estimate for the given option split, or `undefined` on miss. */
|
|
74
|
-
export function readEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean): number | undefined {
|
|
75
|
-
return (excludeEncryptedReasoning ? estimateCacheFloored : estimateCacheDefault).get(message);
|
|
76
|
-
}
|
|
77
|
-
|
|
78
|
-
/** Store an estimate for the given option split. */
|
|
79
|
-
export function writeEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean, value: number): void {
|
|
80
|
-
(excludeEncryptedReasoning ? estimateCacheFloored : estimateCacheDefault).set(message, value);
|
|
81
|
-
}
|
|
82
|
-
|
|
83
82
|
/**
|
|
84
83
|
* Drop every cached derivation of `message` after an in-place rewrite. Owners of
|
|
85
84
|
* mutation (prune, shake, strip-images) call this at the mutation seam so the
|
|
86
85
|
* next convert/estimate pass recomputes from the new content.
|
|
87
86
|
*/
|
|
88
87
|
export function invalidateMessageCache(message: AgentMessage): void {
|
|
89
|
-
|
|
90
|
-
|
|
88
|
+
const versioned = message as VersionedMessage;
|
|
89
|
+
versioned[kEstimateVersion] = ((versioned[kEstimateVersion] ?? 0) + 1) | 0;
|
|
91
90
|
for (const invalidate of externalInvalidators) invalidate(message);
|
|
92
91
|
}
|
|
@@ -49,6 +49,10 @@ export interface CompactionSummaryMessage {
|
|
|
49
49
|
summary: string;
|
|
50
50
|
shortSummary?: string;
|
|
51
51
|
tokensBefore: number;
|
|
52
|
+
/** Estimated context tokens after the rewrite (display metadata). */
|
|
53
|
+
tokensAfter?: number;
|
|
54
|
+
/** Harness compaction method that produced this summary (display metadata). */
|
|
55
|
+
method?: string;
|
|
52
56
|
providerPayload?: ProviderPayload;
|
|
53
57
|
/** Runtime-only ordered archive blocks for snapcompact: old text region,
|
|
54
58
|
* imaged middle, then new text region. When present, `summary` is already
|
|
@@ -99,16 +103,26 @@ export function createBranchSummaryMessage(summary: string, fromId: string, time
|
|
|
99
103
|
};
|
|
100
104
|
}
|
|
101
105
|
|
|
106
|
+
/** Optional metadata for {@link createCompactionSummaryMessage}. */
|
|
107
|
+
export interface CompactionSummaryMessageOptions {
|
|
108
|
+
shortSummary?: string;
|
|
109
|
+
providerPayload?: ProviderPayload;
|
|
110
|
+
images?: ImageContent[];
|
|
111
|
+
blocks?: (TextContent | ImageContent)[];
|
|
112
|
+
warning?: string;
|
|
113
|
+
/** Harness compaction method that produced this summary (e.g. "remote", "soft", "handoff"). */
|
|
114
|
+
method?: string;
|
|
115
|
+
/** Estimated context tokens after the rewrite, for display alongside `tokensBefore`. */
|
|
116
|
+
tokensAfter?: number;
|
|
117
|
+
}
|
|
118
|
+
|
|
102
119
|
export function createCompactionSummaryMessage(
|
|
103
120
|
summary: string,
|
|
104
121
|
tokensBefore: number,
|
|
105
122
|
timestamp: string,
|
|
106
|
-
|
|
107
|
-
providerPayload?: ProviderPayload,
|
|
108
|
-
images?: ImageContent[],
|
|
109
|
-
blocks?: (TextContent | ImageContent)[],
|
|
110
|
-
warning?: string,
|
|
123
|
+
options: CompactionSummaryMessageOptions = {},
|
|
111
124
|
): CompactionSummaryMessage {
|
|
125
|
+
const { shortSummary, providerPayload, images, blocks, warning, method, tokensAfter } = options;
|
|
112
126
|
const imageBlocks =
|
|
113
127
|
blocks?.filter((block): block is ImageContent => block.type === "image") ??
|
|
114
128
|
(images && images.length > 0 ? images : undefined);
|
|
@@ -117,6 +131,8 @@ export function createCompactionSummaryMessage(
|
|
|
117
131
|
summary,
|
|
118
132
|
shortSummary,
|
|
119
133
|
tokensBefore,
|
|
134
|
+
tokensAfter,
|
|
135
|
+
method,
|
|
120
136
|
providerPayload,
|
|
121
137
|
blocks: blocks && blocks.length > 0 ? blocks : undefined,
|
|
122
138
|
images: imageBlocks && imageBlocks.length > 0 ? imageBlocks : undefined,
|
package/src/compaction/openai.ts
CHANGED
|
@@ -51,7 +51,7 @@ import {
|
|
|
51
51
|
OPENAI_HEADERS,
|
|
52
52
|
} from "@oh-my-pi/pi-catalog/wire/codex";
|
|
53
53
|
import { $env, isRecord, logger, prompt, stringifyJson, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
|
54
|
-
import {
|
|
54
|
+
import { Tokenizer } from "../tokenizer";
|
|
55
55
|
import contextWindowTruncatedOutputPrompt from "./prompts/context-window-truncated-output.md" with { type: "text" };
|
|
56
56
|
|
|
57
57
|
export * from "./compaction-v2-streaming";
|
|
@@ -123,14 +123,36 @@ export interface TrimRemoteCompactionInputResult {
|
|
|
123
123
|
estimatedTokensAfter: number;
|
|
124
124
|
}
|
|
125
125
|
|
|
126
|
-
|
|
126
|
+
/** Verdict for one remote-compaction request measured against the model window. */
|
|
127
|
+
interface RemoteCompactionBudgetProbe {
|
|
128
|
+
/** Estimated request tokens; the text part is exact when the cheap bound busted. */
|
|
129
|
+
tokens: number;
|
|
130
|
+
/** Whether the request fits the window. Always true when no window is known. */
|
|
131
|
+
fits: boolean;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Cheap-first sizing of a remote-compaction request. Images and the request
|
|
136
|
+
* frame are charged flat, so they come off the budget rather than through the
|
|
137
|
+
* tokenizer; the serialized transcript is then probed with
|
|
138
|
+
* {@link Tokenizer.checkTokenBudget}, which only pays for an exact count when
|
|
139
|
+
* the byte bound cannot already prove the request fits.
|
|
140
|
+
*/
|
|
141
|
+
function probeRemoteCompactionInputBudget(
|
|
127
142
|
input: Array<Record<string, unknown>>,
|
|
143
|
+
tokenizer: Tokenizer,
|
|
128
144
|
instructions: string,
|
|
129
|
-
tools
|
|
130
|
-
|
|
145
|
+
tools: unknown[] | undefined,
|
|
146
|
+
contextWindow: number | null | undefined,
|
|
147
|
+
): RemoteCompactionBudgetProbe {
|
|
131
148
|
const normalized = normalizeRemoteCompactionEstimateValue({ instructions, input, ...(tools ? { tools } : {}) });
|
|
132
149
|
const serialized = stringifyJson(normalized.value) ?? "";
|
|
133
|
-
|
|
150
|
+
const flatTokens = normalized.imageTokens + REMOTE_COMPACTION_REQUEST_OVERHEAD_TOKENS;
|
|
151
|
+
if (!contextWindow || contextWindow <= 0) {
|
|
152
|
+
return { tokens: tokenizer.countTokens(serialized, "upperbound") + flatTokens, fits: true };
|
|
153
|
+
}
|
|
154
|
+
const budget = tokenizer.checkTokenBudget(serialized, Math.max(0, contextWindow - flatTokens));
|
|
155
|
+
return { tokens: budget.tokens + flatTokens, fits: budget.fits };
|
|
134
156
|
}
|
|
135
157
|
|
|
136
158
|
function rewriteToolOutputForContextWindow(item: Record<string, unknown>): Record<string, unknown> | undefined {
|
|
@@ -164,24 +186,25 @@ function isToolResultImageAttachment(item: Record<string, unknown>): boolean {
|
|
|
164
186
|
*/
|
|
165
187
|
export function trimRemoteCompactionInputToContextWindow(
|
|
166
188
|
input: Array<Record<string, unknown>>,
|
|
189
|
+
tokenizer: Tokenizer,
|
|
167
190
|
contextWindow: number | null | undefined,
|
|
168
191
|
instructions: string,
|
|
169
192
|
tools?: unknown[],
|
|
170
193
|
): TrimRemoteCompactionInputResult {
|
|
171
|
-
const
|
|
172
|
-
if (
|
|
194
|
+
const before = probeRemoteCompactionInputBudget(input, tokenizer, instructions, tools, contextWindow);
|
|
195
|
+
if (before.fits) {
|
|
173
196
|
return {
|
|
174
197
|
input,
|
|
175
198
|
rewrittenOutputs: 0,
|
|
176
|
-
estimatedTokensBefore,
|
|
177
|
-
estimatedTokensAfter:
|
|
199
|
+
estimatedTokensBefore: before.tokens,
|
|
200
|
+
estimatedTokensAfter: before.tokens,
|
|
178
201
|
};
|
|
179
202
|
}
|
|
180
203
|
|
|
181
204
|
let rewrittenInput: Array<Record<string, unknown>> | undefined;
|
|
182
|
-
let
|
|
205
|
+
let after = before;
|
|
183
206
|
let rewrittenOutputs = 0;
|
|
184
|
-
for (let index = input.length - 1; index >= 0 &&
|
|
207
|
+
for (let index = input.length - 1; index >= 0 && !after.fits; index--) {
|
|
185
208
|
const item = input[index];
|
|
186
209
|
if (isToolResultImageAttachment(item)) continue;
|
|
187
210
|
const rewritten = rewriteToolOutputForContextWindow(item);
|
|
@@ -189,23 +212,23 @@ export function trimRemoteCompactionInputToContextWindow(
|
|
|
189
212
|
rewrittenInput ??= input.slice();
|
|
190
213
|
rewrittenInput[index] = rewritten;
|
|
191
214
|
rewrittenOutputs++;
|
|
192
|
-
|
|
215
|
+
after = probeRemoteCompactionInputBudget(rewrittenInput, tokenizer, instructions, tools, contextWindow);
|
|
193
216
|
}
|
|
194
217
|
|
|
195
|
-
if (!rewrittenInput ||
|
|
218
|
+
if (!rewrittenInput || !after.fits) {
|
|
196
219
|
return {
|
|
197
220
|
input,
|
|
198
221
|
rewrittenOutputs: 0,
|
|
199
|
-
estimatedTokensBefore,
|
|
200
|
-
estimatedTokensAfter:
|
|
222
|
+
estimatedTokensBefore: before.tokens,
|
|
223
|
+
estimatedTokensAfter: before.tokens,
|
|
201
224
|
};
|
|
202
225
|
}
|
|
203
226
|
|
|
204
227
|
return {
|
|
205
228
|
input: rewrittenInput,
|
|
206
229
|
rewrittenOutputs,
|
|
207
|
-
estimatedTokensBefore,
|
|
208
|
-
estimatedTokensAfter,
|
|
230
|
+
estimatedTokensBefore: before.tokens,
|
|
231
|
+
estimatedTokensAfter: after.tokens,
|
|
209
232
|
};
|
|
210
233
|
}
|
|
211
234
|
|
|
@@ -766,7 +789,12 @@ export async function requestOpenAiRemoteCompaction(
|
|
|
766
789
|
): Promise<OpenAiRemoteCompactionResponse> {
|
|
767
790
|
const endpoint = resolveOpenAiCompactEndpoint(model);
|
|
768
791
|
const requestModel = resolveOpenAiCompactModel(model);
|
|
769
|
-
const trimmed = trimRemoteCompactionInputToContextWindow(
|
|
792
|
+
const trimmed = trimRemoteCompactionInputToContextWindow(
|
|
793
|
+
compactInput,
|
|
794
|
+
new Tokenizer(model),
|
|
795
|
+
model.contextWindow,
|
|
796
|
+
instructions,
|
|
797
|
+
);
|
|
770
798
|
if (trimmed.rewrittenOutputs > 0) {
|
|
771
799
|
logger.info("Rewrote trailing tool outputs before OpenAI remote compaction", {
|
|
772
800
|
model: model.id,
|