@oh-my-pi/pi-agent-core 17.3.8 → 17.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -31,11 +31,11 @@ import { buildResponsesInput, resolveOpenAICompatPolicy } from "@oh-my-pi/pi-ai/
31
31
  import { stripOpenAIResponsesOutputOnlyStatusesForReplay } from "@oh-my-pi/pi-ai/utils";
32
32
  import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
33
33
  import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking";
34
- import { isRecord, logger, prompt, stringifyJson } from "@oh-my-pi/pi-utils";
34
+ import { isRecord, logger, prompt } from "@oh-my-pi/pi-utils";
35
35
  import * as snapcompact from "@oh-my-pi/snapcompact";
36
36
  import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
37
37
  import { ThinkingLevel } from "../thinking";
38
- import { countTokens } from "../tokenizer";
38
+ import { Tokenizer } from "../tokenizer";
39
39
  import type { AgentMessage } from "../types";
40
40
  import {
41
41
  buildCompactionV2Request,
@@ -47,7 +47,6 @@ import {
47
47
  } from "./compaction-v2-streaming";
48
48
  import type { CompactionEntry, SessionEntry } from "./entries";
49
49
  import { NativeCompactionError } from "./errors";
50
- import { isEstimateCacheable, readEstimateCache, writeEstimateCache } from "./message-cache";
51
50
  import { type ConvertToLlm, createBranchSummaryMessage, createCustomMessage, defaultConvertToLlm } from "./messages";
52
51
  import {
53
52
  buildOpenAiNativeHistory,
@@ -390,144 +389,17 @@ export function resolveThresholdTokens(contextWindow: number, settings: Compacti
390
389
  // Cut point detection
391
390
  // ============================================================================
392
391
 
393
- /**
394
- * Image content has no tokenizer representation; charge a fixed estimate
395
- * matching what providers typically bill for inline images.
396
- */
397
- const IMAGE_TOKEN_ESTIMATE = 1200;
398
-
399
- /**
400
- * Estimate token count for a message using cl100k_base via the native
401
- * tokenizer. This is not Claude's first-party tokenizer (Anthropic doesn't
402
- * publish one) but is within ~5–10% across English/code text.
403
- *
404
- * `excludeEncryptedReasoning` drops opaque provider reasoning payloads
405
- * (`thinkingSignature`, `redactedThinking`) from the estimate. Those are billed
406
- * by the provider on replay, so the default counts them — but their *local*
407
- * byte size can diverge wildly from what the provider charges, so the
408
- * compaction floor (which only needs the reliably-countable, on-wire-compressible
409
- * content) excludes them to avoid false triggers on thinking-heavy turns.
410
- */
411
- export function estimateTokens(message: AgentMessage, options?: { excludeEncryptedReasoning?: boolean }): number {
412
- // Settled historical messages are counted once and reused until an owner
413
- // (prune/shake/strip-images) invalidates them; streaming assistants bypass
414
- // the cache entirely (see message-cache.ts settle-gate invariant).
415
- const cacheable = isEstimateCacheable(message);
416
- const excludeEncryptedReasoning = options?.excludeEncryptedReasoning === true;
417
- if (cacheable) {
418
- const cached = readEstimateCache(message, excludeEncryptedReasoning);
419
- if (cached !== undefined) return cached;
420
- }
421
- const result = computeMessageTokens(message, options);
422
- if (cacheable) writeEstimateCache(message, excludeEncryptedReasoning, result);
423
- return result;
424
- }
425
-
426
- function computeMessageTokens(message: AgentMessage, options?: { excludeEncryptedReasoning?: boolean }): number {
427
- const fragments: string[] = [];
428
- let extra = 0;
429
- if ((message as { role?: string }).role === "bashExecution") {
430
- const bash = message as { command?: unknown; output?: unknown };
431
- if (typeof bash.command === "string") fragments.push(bash.command);
432
- if (typeof bash.output === "string") fragments.push(bash.output);
433
- return fragments.length === 0 ? 0 : countTokens(fragments);
434
- }
435
-
436
- switch (message.role) {
437
- case "user": {
438
- const content = (message as { content: string | Array<{ type: string; text?: string }> }).content;
439
- if (typeof content === "string") {
440
- fragments.push(content);
441
- } else if (Array.isArray(content)) {
442
- for (const block of content) {
443
- if (block.type === "text" && block.text) {
444
- fragments.push(block.text);
445
- }
446
- }
447
- }
448
- break;
449
- }
450
- case "assistant": {
451
- const assistant = message as AssistantMessage;
452
- for (const block of assistant.content) {
453
- if (block.type === "text") {
454
- fragments.push(block.text);
455
- } else if (block.type === "thinking") {
456
- fragments.push(block.thinking);
457
- // Providers charge for the opaque signature/reasoning payload that
458
- // rides alongside the thinking text (OpenAI Responses encrypted
459
- // reasoning items, Anthropic signed thinking blocks, etc.). Without
460
- // counting it, this estimator can read ~half of the provider-reported
461
- // usage on thinking-heavy turns — see #2275 for the resulting
462
- // compaction-trigger / post-check metric divergence. The compaction
463
- // floor excludes it (its local byte size diverges from provider billing).
464
- if (block.thinkingSignature && !options?.excludeEncryptedReasoning) {
465
- fragments.push(block.thinkingSignature);
466
- }
467
- } else if (block.type === "toolCall") {
468
- fragments.push(block.name);
469
- fragments.push(stringifyJson(block.arguments) ?? "null");
470
- } else if (block.type === "redactedThinking") {
471
- // Encrypted reasoning blob the provider still bills for on replay;
472
- // excluded from the compaction floor for the same reason as above.
473
- if (!options?.excludeEncryptedReasoning) fragments.push(block.data);
474
- } else if (block.type === "anthropicServerTool") {
475
- // Native Anthropic server-tool call/result replayed verbatim on the
476
- // wire (server_tool_use input and opaque result content). This opaque
477
- // provider-replay state the provider still
478
- // bills for on same-provider replay; excluded from the compaction
479
- // floor like other encrypted reasoning because its local byte size
480
- // diverges from provider billing.
481
- if (!options?.excludeEncryptedReasoning) fragments.push(stringifyJson(block.block) ?? "null");
482
- }
483
- }
484
- break;
485
- }
486
- case "hookMessage":
487
- case "toolResult": {
488
- if (typeof message.content === "string") {
489
- fragments.push(message.content);
490
- } else {
491
- for (const block of message.content) {
492
- if (block.type === "text" && block.text) {
493
- fragments.push(block.text);
494
- } else if (block.type === "image") {
495
- extra += IMAGE_TOKEN_ESTIMATE;
496
- }
497
- }
498
- }
499
- break;
500
- }
501
- case "branchSummary":
502
- case "compactionSummary": {
503
- fragments.push(message.summary);
504
- if (message.role === "compactionSummary") {
505
- if (message.blocks) {
506
- for (const block of message.blocks) {
507
- if (block.type === "text") fragments.push(block.text);
508
- else extra += snapcompact.FRAME_TOKEN_ESTIMATE;
509
- }
510
- } else if (message.images) {
511
- // Snapcompact frames render at ≥1568px; providers bill the downscaled cap.
512
- extra += message.images.length * snapcompact.FRAME_TOKEN_ESTIMATE;
513
- }
514
- }
515
- break;
516
- }
517
- default:
518
- return 0;
519
- }
520
-
521
- if (fragments.length === 0) return extra;
522
- return extra + countTokens(fragments);
523
- }
524
-
525
- function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number {
392
+ function estimateEntriesTokens(
393
+ entries: SessionEntry[],
394
+ tokenizer: Tokenizer,
395
+ startIndex: number,
396
+ endIndex: number,
397
+ ): number {
526
398
  let total = 0;
527
399
  for (let i = startIndex; i < endIndex; i++) {
528
400
  const msg = getMessageFromEntry(entries[i]);
529
401
  if (msg) {
530
- total += estimateTokens(msg);
402
+ total += tokenizer.countMessage(msg);
531
403
  }
532
404
  }
533
405
  return total;
@@ -626,6 +498,7 @@ export interface CutPointResult {
626
498
  */
627
499
  export function findCutPoint(
628
500
  entries: SessionEntry[],
501
+ tokenizer: Tokenizer,
629
502
  startIndex: number,
630
503
  endIndex: number,
631
504
  keepRecentTokens: number,
@@ -645,7 +518,7 @@ export function findCutPoint(
645
518
  if (entry.type !== "message") continue;
646
519
 
647
520
  // Estimate this message's size
648
- const messageTokens = estimateTokens(entry.message);
521
+ const messageTokens = tokenizer.countMessage(entry.message);
649
522
  accumulatedTokens += messageTokens;
650
523
 
651
524
  // Check if we've exceeded the budget
@@ -948,12 +821,17 @@ interface SummaryWindow {
948
821
  * on message boundaries. Only called when the whole conversation does not fit —
949
822
  * the common single-window path never pays this per-message sizing pass.
950
823
  */
951
- function planSummaryWindows(messages: Message[], dialect: Dialect | undefined, budgetTokens: number): Message[][] {
824
+ function planSummaryWindows(
825
+ messages: Message[],
826
+ tokenizer: Tokenizer,
827
+ dialect: Dialect | undefined,
828
+ budgetTokens: number,
829
+ ): Message[][] {
952
830
  const windows: Message[][] = [];
953
831
  let current: Message[] = [];
954
832
  let currentTokens = 0;
955
833
  for (const message of messages) {
956
- const tokens = countTokens(serializeConversationForSummary([message], dialect));
834
+ const tokens = tokenizer.countTokens(serializeConversationForSummary([message], dialect));
957
835
  if (currentTokens > 0 && currentTokens + tokens > budgetTokens) {
958
836
  windows.push(current);
959
837
  current = [];
@@ -982,6 +860,7 @@ export async function generateSummary(
982
860
  // Convert to LLM messages first (handles custom app messages when caller provides a transformer).
983
861
  const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(currentMessages);
984
862
  const dialect = preferredDialect(model.id);
863
+ const tokenizer = new Tokenizer(model);
985
864
  const wholeConversation = serializeConversationForSummary(llmMessages, dialect);
986
865
  const budgetTokens = summaryInputBudgetTokens(model, maxTokens);
987
866
  // A span that outgrew the summarizer's window is summarized as a fold: each
@@ -991,19 +870,21 @@ export async function generateSummary(
991
870
  // retry can shrink — the state a cross-provider compaction boundary
992
871
  // (see `prepareCompaction`) puts a long session into. One window is the
993
872
  // common case and costs exactly the one call it always did.
994
- const pending: SummaryWindow[] =
995
- countTokens(wholeConversation) <= budgetTokens
996
- ? [{ messages: llmMessages, budgetTokens, text: wholeConversation }]
997
- : planSummaryWindows(llmMessages, dialect, budgetTokens).map(messages => ({ messages, budgetTokens }));
873
+ const pending: SummaryWindow[] = tokenizer.checkTokenBudget(wholeConversation, budgetTokens).fits
874
+ ? [{ messages: llmMessages, budgetTokens, text: wholeConversation }]
875
+ : planSummaryWindows(llmMessages, tokenizer, dialect, budgetTokens).map(messages => ({ messages, budgetTokens }));
998
876
 
999
877
  let carriedSummary = previousSummary;
1000
878
  while (pending.length > 0) {
1001
879
  const window = pending[0];
1002
880
  const text = window.text ?? serializeConversationForSummary(window.messages, dialect);
1003
- const windowTokens = countTokens(text);
881
+ // A budget probe, not a raw count: a window whose bytes already fit needs
882
+ // neither an exact count nor the clamp, and the bust path hands back the
883
+ // exact count the proportional clamp needs as its denominator.
884
+ const budget = tokenizer.checkTokenBudget(text, window.budgetTokens);
1004
885
  try {
1005
886
  carriedSummary = await summarizeConversationWindow(
1006
- clampConversationToBudget(text, window.budgetTokens, windowTokens),
887
+ budget.fits ? text : clampConversationToBudget(text, window.budgetTokens, budget.tokens),
1007
888
  carriedSummary,
1008
889
  model,
1009
890
  maxTokens,
@@ -1020,8 +901,11 @@ export async function generateSummary(
1020
901
  // window size only the provider can tell us is wrong.
1021
902
  // Halve what was actually SENT, not the budget it was planned against:
1022
903
  // the rejection proves the plan was fiction, so converging on the real
1023
- // cap must not spend a call per level of an imaginary ladder.
1024
- const halved = Math.floor(Math.min(window.budgetTokens, windowTokens) / 2);
904
+ // cap must not spend a call per level of an imaginary ladder. The cheap
905
+ // fit path never counted this window, so pay for the exact size here —
906
+ // one tokenization is nothing against the provider round trip already lost.
907
+ const sentTokens = budget.exact ? budget.tokens : tokenizer.countTokens(text, "strict");
908
+ const halved = Math.floor(Math.min(window.budgetTokens, sentTokens) / 2);
1025
909
  if (
1026
910
  !AIError.is(AIError.classify(error), AIError.Flag.ContextOverflow) ||
1027
911
  halved < minSummaryInputTokens(model)
@@ -1031,7 +915,7 @@ export async function generateSummary(
1031
915
  pending.splice(
1032
916
  0,
1033
917
  1,
1034
- ...planSummaryWindows(window.messages, dialect, halved).map(messages => ({
918
+ ...planSummaryWindows(window.messages, tokenizer, dialect, halved).map(messages => ({
1035
919
  messages,
1036
920
  budgetTokens: halved,
1037
921
  })),
@@ -1384,7 +1268,7 @@ export interface CompactionPreparation {
1384
1268
  * let the active model replay it, so keying reuse on "any candidate shares the
1385
1269
  * provider" left a provider-switched session permanently context-less (#6343).
1386
1270
  */
1387
- function remotePreserveReusable(
1271
+ export function remotePreserveReusable(
1388
1272
  preserveData: Record<string, unknown> | undefined,
1389
1273
  activeModel: Model,
1390
1274
  settings: CompactionSettings,
@@ -1423,10 +1307,16 @@ export function findReadableCompactionIndex(
1423
1307
  return -1;
1424
1308
  }
1425
1309
 
1310
+ /**
1311
+ * Pass the caller's warm `tokenizer` (the Agent's for the active model) so the
1312
+ * full-branch estimate walk hits its memo; the cold default is for one-shot
1313
+ * callers that have no live agent.
1314
+ */
1426
1315
  export function prepareCompaction(
1427
1316
  pathEntries: SessionEntry[],
1428
1317
  settings: CompactionSettings,
1429
1318
  activeModel?: Model,
1319
+ tokenizer: Tokenizer = new Tokenizer(activeModel),
1430
1320
  ): CompactionPreparation | undefined {
1431
1321
  if (pathEntries.length > 0 && pathEntries[pathEntries.length - 1].type === "compaction") {
1432
1322
  return undefined;
@@ -1459,7 +1349,7 @@ export function prepareCompaction(
1459
1349
  const tokensBefore = lastUsage ? calculateContextTokens(lastUsage) : 0;
1460
1350
  let keepRecentTokens = settings.keepRecentTokens;
1461
1351
  if (lastUsage) {
1462
- const estimatedTokens = estimateEntriesTokens(pathEntries, boundaryStart, boundaryEnd);
1352
+ const estimatedTokens = estimateEntriesTokens(pathEntries, tokenizer, boundaryStart, boundaryEnd);
1463
1353
  const promptTokens = calculatePromptTokens(lastUsage);
1464
1354
  const ratio = estimatedTokens > 0 ? promptTokens / estimatedTokens : 0;
1465
1355
  if (Number.isFinite(ratio) && ratio > 1) {
@@ -1467,7 +1357,7 @@ export function prepareCompaction(
1467
1357
  }
1468
1358
  }
1469
1359
 
1470
- const cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, keepRecentTokens);
1360
+ const cutPoint = findCutPoint(pathEntries, tokenizer, boundaryStart, boundaryEnd, keepRecentTokens);
1471
1361
 
1472
1362
  // Get ID of first kept entry
1473
1363
  const firstKeptEntry = pathEntries[cutPoint.firstKeptEntryIndex];
@@ -1706,6 +1596,7 @@ export async function compact(
1706
1596
  : undefined;
1707
1597
  const trimmed = trimRemoteCompactionInputToContextWindow(
1708
1598
  remoteHistory,
1599
+ new Tokenizer(model),
1709
1600
  model.contextWindow,
1710
1601
  instructions,
1711
1602
  tools,
@@ -11,4 +11,5 @@ export * from "./messages";
11
11
  export * from "./openai";
12
12
  export * from "./pruning";
13
13
  export * from "./shake";
14
+ export * from "./transcript-tokens";
14
15
  export * from "./utils";
@@ -1,13 +1,12 @@
1
1
  /**
2
- * Per-message memoization for the two hot history walks: token estimation
3
- * ({@link estimateTokens}) and LLM conversion (the coding-agent's `convertToLlm`).
2
+ * Cache-coherence seams for the two hot history walks: token estimation
3
+ * ({@link Tokenizer.countMessage}) and LLM conversion (the coding-agent's
4
+ * `convertToLlm`).
4
5
  *
5
6
  * Long sessions re-walk a settled `AgentMessage[]` every turn, re-tokenizing and
6
- * re-converting historical objects that only the newest suffix can change. These
7
- * caches key on message *identity* so a settled message is counted/converted once
8
- * and reused until an owner rewrites it.
9
- *
10
- * Correctness rests on two invariants:
7
+ * re-converting historical objects that only the newest suffix can change. Each
8
+ * `Tokenizer` memoizes estimates per message identity; this module owns the two
9
+ * invariants that keep those memos (and the cross-package convert memo) honest:
11
10
  *
12
11
  * 1. **Settle gate.** A streaming assistant is mutated under one identity while
13
12
  * its `usage`/`stopReason` are provisional (the seed carries zeroed usage and
@@ -19,9 +18,11 @@
19
18
  * 2. **Owner invalidation.** `pruneToolOutputs` / `pruneSupersededToolResults`,
20
19
  * `applyShakeRegion`, and `stripImagesFromMessage` rewrite message content in
21
20
  * place under a stable identity. Each MUST call {@link invalidateMessageCache}
22
- * on the mutated message before the next convert/estimate pass so both caches
23
- * drop the stale entry. The convert cache lives in another package, so it
24
- * subscribes via {@link registerMessageCacheInvalidator}.
21
+ * on the mutated message before the next convert/estimate pass. Invalidation
22
+ * bumps a symbol-keyed version tag on the message itself, so every live
23
+ * `Tokenizer` memo drops its stale entry at once without registering
24
+ * anywhere; the convert cache lives in another package and subscribes via
25
+ * {@link registerMessageCacheInvalidator}.
25
26
  */
26
27
  import type { AssistantMessage } from "@oh-my-pi/pi-ai";
27
28
  import type { AgentMessage } from "../types";
@@ -41,18 +42,26 @@ export function registerMessageCacheInvalidator(invalidate: (message: AgentMessa
41
42
  };
42
43
  }
43
44
 
44
- // Dual option-split estimate caches: the compaction floor passes
45
- // `excludeEncryptedReasoning` (dropping opaque provider reasoning), so a message
46
- // has two distinct estimates that must not collide in one map.
47
- //
48
- // These are WeakMaps, not symbol-tagged properties, deliberately: callers spread
49
- // messages to derive throwaway variants for counting — `estimateBranchSummaryTokens`
50
- // does `estimateTokens({ ...message, content: truncated })`. A symbol-keyed cache
51
- // value rides along an object spread, so the truncated clone would inherit (and
52
- // return) the full-content estimate. Keying strictly on identity keeps the cache
53
- // off spread copies, which get their own fresh count.
54
- const estimateCacheDefault = new WeakMap<AgentMessage, number>();
55
- const estimateCacheFloored = new WeakMap<AgentMessage, number>();
45
+ /**
46
+ * Estimate-version tag riding on the message itself. Symbol-keyed, so JSON
47
+ * session persistence and default iteration never see it. Object spread copies
48
+ * the tag onto derived clones — harmless, because estimate memos key on message
49
+ * *identity* and a fresh clone starts with no memo entries anywhere.
50
+ */
51
+ const kEstimateVersion = Symbol("omp.messageEstimateVersion");
52
+
53
+ interface VersionedMessage {
54
+ [kEstimateVersion]?: number;
55
+ }
56
+
57
+ /**
58
+ * Current estimate version of `message` (0 until first invalidation). A
59
+ * `Tokenizer` memo entry stamped with an older version is stale and must be
60
+ * recounted.
61
+ */
62
+ export function messageEstimateVersion(message: AgentMessage): number {
63
+ return (message as VersionedMessage)[kEstimateVersion] ?? 0;
64
+ }
56
65
 
57
66
  /**
58
67
  * True when this message's estimate is safe to cache by identity. Non-assistants
@@ -70,23 +79,13 @@ export function isEstimateCacheable(message: AgentMessage): boolean {
70
79
  );
71
80
  }
72
81
 
73
- /** Read a cached estimate for the given option split, or `undefined` on miss. */
74
- export function readEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean): number | undefined {
75
- return (excludeEncryptedReasoning ? estimateCacheFloored : estimateCacheDefault).get(message);
76
- }
77
-
78
- /** Store an estimate for the given option split. */
79
- export function writeEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean, value: number): void {
80
- (excludeEncryptedReasoning ? estimateCacheFloored : estimateCacheDefault).set(message, value);
81
- }
82
-
83
82
  /**
84
83
  * Drop every cached derivation of `message` after an in-place rewrite. Owners of
85
84
  * mutation (prune, shake, strip-images) call this at the mutation seam so the
86
85
  * next convert/estimate pass recomputes from the new content.
87
86
  */
88
87
  export function invalidateMessageCache(message: AgentMessage): void {
89
- estimateCacheDefault.delete(message);
90
- estimateCacheFloored.delete(message);
88
+ const versioned = message as VersionedMessage;
89
+ versioned[kEstimateVersion] = ((versioned[kEstimateVersion] ?? 0) + 1) | 0;
91
90
  for (const invalidate of externalInvalidators) invalidate(message);
92
91
  }
@@ -49,6 +49,10 @@ export interface CompactionSummaryMessage {
49
49
  summary: string;
50
50
  shortSummary?: string;
51
51
  tokensBefore: number;
52
+ /** Estimated context tokens after the rewrite (display metadata). */
53
+ tokensAfter?: number;
54
+ /** Harness compaction method that produced this summary (display metadata). */
55
+ method?: string;
52
56
  providerPayload?: ProviderPayload;
53
57
  /** Runtime-only ordered archive blocks for snapcompact: old text region,
54
58
  * imaged middle, then new text region. When present, `summary` is already
@@ -99,16 +103,26 @@ export function createBranchSummaryMessage(summary: string, fromId: string, time
99
103
  };
100
104
  }
101
105
 
106
+ /** Optional metadata for {@link createCompactionSummaryMessage}. */
107
+ export interface CompactionSummaryMessageOptions {
108
+ shortSummary?: string;
109
+ providerPayload?: ProviderPayload;
110
+ images?: ImageContent[];
111
+ blocks?: (TextContent | ImageContent)[];
112
+ warning?: string;
113
+ /** Harness compaction method that produced this summary (e.g. "remote", "soft", "handoff"). */
114
+ method?: string;
115
+ /** Estimated context tokens after the rewrite, for display alongside `tokensBefore`. */
116
+ tokensAfter?: number;
117
+ }
118
+
102
119
  export function createCompactionSummaryMessage(
103
120
  summary: string,
104
121
  tokensBefore: number,
105
122
  timestamp: string,
106
- shortSummary?: string,
107
- providerPayload?: ProviderPayload,
108
- images?: ImageContent[],
109
- blocks?: (TextContent | ImageContent)[],
110
- warning?: string,
123
+ options: CompactionSummaryMessageOptions = {},
111
124
  ): CompactionSummaryMessage {
125
+ const { shortSummary, providerPayload, images, blocks, warning, method, tokensAfter } = options;
112
126
  const imageBlocks =
113
127
  blocks?.filter((block): block is ImageContent => block.type === "image") ??
114
128
  (images && images.length > 0 ? images : undefined);
@@ -117,6 +131,8 @@ export function createCompactionSummaryMessage(
117
131
  summary,
118
132
  shortSummary,
119
133
  tokensBefore,
134
+ tokensAfter,
135
+ method,
120
136
  providerPayload,
121
137
  blocks: blocks && blocks.length > 0 ? blocks : undefined,
122
138
  images: imageBlocks && imageBlocks.length > 0 ? imageBlocks : undefined,
@@ -45,13 +45,14 @@ import {
45
45
  } from "@oh-my-pi/pi-ai/utils";
46
46
  import { captureOpenAIHttpError } from "@oh-my-pi/pi-ai/utils/openai-http";
47
47
  import {
48
+ applyCodexResidencyHeader,
48
49
  CODEX_BASE_URL,
49
50
  getCodexAccountId,
50
51
  OPENAI_HEADER_VALUES,
51
52
  OPENAI_HEADERS,
52
53
  } from "@oh-my-pi/pi-catalog/wire/codex";
53
54
  import { $env, isRecord, logger, prompt, stringifyJson, structuredCloneJSON } from "@oh-my-pi/pi-utils";
54
- import { countTokensConservatively } from "../tokenizer";
55
+ import { Tokenizer } from "../tokenizer";
55
56
  import contextWindowTruncatedOutputPrompt from "./prompts/context-window-truncated-output.md" with { type: "text" };
56
57
 
57
58
  export * from "./compaction-v2-streaming";
@@ -123,14 +124,36 @@ export interface TrimRemoteCompactionInputResult {
123
124
  estimatedTokensAfter: number;
124
125
  }
125
126
 
126
- function estimateRemoteCompactionInputTokens(
127
+ /** Verdict for one remote-compaction request measured against the model window. */
128
+ interface RemoteCompactionBudgetProbe {
129
+ /** Estimated request tokens; the text part is exact when the cheap bound busted. */
130
+ tokens: number;
131
+ /** Whether the request fits the window. Always true when no window is known. */
132
+ fits: boolean;
133
+ }
134
+
135
+ /**
136
+ * Cheap-first sizing of a remote-compaction request. Images and the request
137
+ * frame are charged flat, so they come off the budget rather than through the
138
+ * tokenizer; the serialized transcript is then probed with
139
+ * {@link Tokenizer.checkTokenBudget}, which only pays for an exact count when
140
+ * the byte bound cannot already prove the request fits.
141
+ */
142
+ function probeRemoteCompactionInputBudget(
127
143
  input: Array<Record<string, unknown>>,
144
+ tokenizer: Tokenizer,
128
145
  instructions: string,
129
- tools?: unknown[],
130
- ): number {
146
+ tools: unknown[] | undefined,
147
+ contextWindow: number | null | undefined,
148
+ ): RemoteCompactionBudgetProbe {
131
149
  const normalized = normalizeRemoteCompactionEstimateValue({ instructions, input, ...(tools ? { tools } : {}) });
132
150
  const serialized = stringifyJson(normalized.value) ?? "";
133
- return countTokensConservatively(serialized) + normalized.imageTokens + REMOTE_COMPACTION_REQUEST_OVERHEAD_TOKENS;
151
+ const flatTokens = normalized.imageTokens + REMOTE_COMPACTION_REQUEST_OVERHEAD_TOKENS;
152
+ if (!contextWindow || contextWindow <= 0) {
153
+ return { tokens: tokenizer.countTokens(serialized, "upperbound") + flatTokens, fits: true };
154
+ }
155
+ const budget = tokenizer.checkTokenBudget(serialized, Math.max(0, contextWindow - flatTokens));
156
+ return { tokens: budget.tokens + flatTokens, fits: budget.fits };
134
157
  }
135
158
 
136
159
  function rewriteToolOutputForContextWindow(item: Record<string, unknown>): Record<string, unknown> | undefined {
@@ -164,24 +187,25 @@ function isToolResultImageAttachment(item: Record<string, unknown>): boolean {
164
187
  */
165
188
  export function trimRemoteCompactionInputToContextWindow(
166
189
  input: Array<Record<string, unknown>>,
190
+ tokenizer: Tokenizer,
167
191
  contextWindow: number | null | undefined,
168
192
  instructions: string,
169
193
  tools?: unknown[],
170
194
  ): TrimRemoteCompactionInputResult {
171
- const estimatedTokensBefore = estimateRemoteCompactionInputTokens(input, instructions, tools);
172
- if (!contextWindow || contextWindow <= 0 || estimatedTokensBefore <= contextWindow) {
195
+ const before = probeRemoteCompactionInputBudget(input, tokenizer, instructions, tools, contextWindow);
196
+ if (before.fits) {
173
197
  return {
174
198
  input,
175
199
  rewrittenOutputs: 0,
176
- estimatedTokensBefore,
177
- estimatedTokensAfter: estimatedTokensBefore,
200
+ estimatedTokensBefore: before.tokens,
201
+ estimatedTokensAfter: before.tokens,
178
202
  };
179
203
  }
180
204
 
181
205
  let rewrittenInput: Array<Record<string, unknown>> | undefined;
182
- let estimatedTokensAfter = estimatedTokensBefore;
206
+ let after = before;
183
207
  let rewrittenOutputs = 0;
184
- for (let index = input.length - 1; index >= 0 && estimatedTokensAfter > contextWindow; index--) {
208
+ for (let index = input.length - 1; index >= 0 && !after.fits; index--) {
185
209
  const item = input[index];
186
210
  if (isToolResultImageAttachment(item)) continue;
187
211
  const rewritten = rewriteToolOutputForContextWindow(item);
@@ -189,23 +213,23 @@ export function trimRemoteCompactionInputToContextWindow(
189
213
  rewrittenInput ??= input.slice();
190
214
  rewrittenInput[index] = rewritten;
191
215
  rewrittenOutputs++;
192
- estimatedTokensAfter = estimateRemoteCompactionInputTokens(rewrittenInput, instructions, tools);
216
+ after = probeRemoteCompactionInputBudget(rewrittenInput, tokenizer, instructions, tools, contextWindow);
193
217
  }
194
218
 
195
- if (!rewrittenInput || estimatedTokensAfter > contextWindow) {
219
+ if (!rewrittenInput || !after.fits) {
196
220
  return {
197
221
  input,
198
222
  rewrittenOutputs: 0,
199
- estimatedTokensBefore,
200
- estimatedTokensAfter: estimatedTokensBefore,
223
+ estimatedTokensBefore: before.tokens,
224
+ estimatedTokensAfter: before.tokens,
201
225
  };
202
226
  }
203
227
 
204
228
  return {
205
229
  input: rewrittenInput,
206
230
  rewrittenOutputs,
207
- estimatedTokensBefore,
208
- estimatedTokensAfter,
231
+ estimatedTokensBefore: before.tokens,
232
+ estimatedTokensAfter: after.tokens,
209
233
  };
210
234
  }
211
235
 
@@ -766,7 +790,12 @@ export async function requestOpenAiRemoteCompaction(
766
790
  ): Promise<OpenAiRemoteCompactionResponse> {
767
791
  const endpoint = resolveOpenAiCompactEndpoint(model);
768
792
  const requestModel = resolveOpenAiCompactModel(model);
769
- const trimmed = trimRemoteCompactionInputToContextWindow(compactInput, model.contextWindow, instructions);
793
+ const trimmed = trimRemoteCompactionInputToContextWindow(
794
+ compactInput,
795
+ new Tokenizer(model),
796
+ model.contextWindow,
797
+ instructions,
798
+ );
770
799
  if (trimmed.rewrittenOutputs > 0) {
771
800
  logger.info("Rewrote trailing tool outputs before OpenAI remote compaction", {
772
801
  model: model.id,
@@ -806,6 +835,7 @@ export async function requestOpenAiRemoteCompaction(
806
835
  if (accountId) {
807
836
  headers[OPENAI_HEADERS.ACCOUNT_ID] = accountId;
808
837
  }
838
+ applyCodexResidencyHeader(headers, apiKey);
809
839
  const attestation = await getCodexAttestationHeader(accountId);
810
840
  if (attestation) {
811
841
  headers[OPENAI_HEADERS.ATTESTATION] = attestation;