@oh-my-pi/pi-agent-core 17.3.8 → 17.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,6 +2,7 @@ import { type ApiKey, type AssistantMessage, type AssistantMessageEvent, type Co
2
2
  import type { Dialect } from "@oh-my-pi/pi-ai/dialect";
3
3
  import type { HarmonyAuditEvent } from "@oh-my-pi/pi-ai/utils/harmony-leak";
4
4
  import type { AppendOnlyContextManager } from "./append-only-context.js";
5
+ import { Tokenizer } from "./tokenizer.js";
5
6
  import type { AgentBeforeModelCall, AgentEvent, AgentLoopConfig, AgentMessage, AgentState, AgentTool, AgentToolContext, AgentTurnEndContext, AsideMessage, StreamFn, ToolCallContext, ToolChoiceDirective } from "./types.js";
6
7
  export declare class AgentBusyError extends Error {
7
8
  constructor(message?: string);
@@ -354,6 +355,12 @@ export declare class Agent {
354
355
  */
355
356
  set maxRetryDelayMs(value: number | undefined);
356
357
  get state(): AgentState;
358
+ /**
359
+ * Tokenizer for the active model. The instance is replaced whenever the
360
+ * active model's encoding changes (see {@link setModel}), so callers must
361
+ * not cache it across model switches.
362
+ */
363
+ get tokenizer(): Tokenizer;
357
364
  get appendOnlyContext(): AppendOnlyContextManager | undefined;
358
365
  setAppendOnlyContext(manager?: AppendOnlyContextManager): void;
359
366
  /**
@@ -401,7 +408,7 @@ export declare class Agent {
401
408
  setAsideMessageProvider(fn: (() => AsideMessage[] | Promise<AsideMessage[]>) | undefined): void;
402
409
  emitExternalEvent(event: AgentEvent): void;
403
410
  setSystemPrompt(v: string[] | string): void;
404
- setModel(m: Model): void;
411
+ setModel(model: Model): void;
405
412
  setThinkingLevel(l: Effort | undefined): void;
406
413
  setDisableReasoning(disabled: boolean): void;
407
414
  setSteeringMode(mode: "all" | "one-at-a-time"): void;
@@ -6,6 +6,7 @@
6
6
  */
7
7
  import type { Api, ApiKey, AssistantMessage, Context, Model, SimpleStreamOptions } from "@oh-my-pi/pi-ai";
8
8
  import { type AgentTelemetry } from "../telemetry.js";
9
+ import { Tokenizer } from "../tokenizer.js";
9
10
  import type { AgentMessage } from "../types.js";
10
11
  import type { ReadonlySessionManager, SessionEntry } from "./entries.js";
11
12
  import { type ConvertToLlm } from "./messages.js";
@@ -91,7 +92,7 @@ export declare function collectEntriesForBranchSummary(session: ReadonlySessionM
91
92
  * @param entries - Entries in chronological order
92
93
  * @param tokenBudget - Maximum tokens to include (0 = no limit)
93
94
  */
94
- export declare function prepareBranchEntries(entries: SessionEntry[], tokenBudget?: number): BranchPreparation;
95
+ export declare function prepareBranchEntries(entries: SessionEntry[], tokenizer: Tokenizer, tokenBudget?: number): BranchPreparation;
95
96
  /**
96
97
  * Generate a summary of abandoned branch entries.
97
98
  *
@@ -7,6 +7,7 @@
7
7
  import { type Api, type ApiKey, type AssistantMessage, type CodexCompactionContext, type Context, type FetchImpl, type MessageAttribution, type Model, type OneshotRetryOptions, type ProviderSessionState, type SimpleStreamOptions, type Tool, type Usage } from "@oh-my-pi/pi-ai";
8
8
  import { type AgentTelemetry } from "../telemetry.js";
9
9
  import { ThinkingLevel } from "../thinking.js";
10
+ import { Tokenizer } from "../tokenizer.js";
10
11
  import type { AgentMessage } from "../types.js";
11
12
  import type { SessionEntry } from "./entries.js";
12
13
  import { type ConvertToLlm } from "./messages.js";
@@ -119,21 +120,6 @@ export declare function shouldCompact(contextTokens: number, contextWindow: numb
119
120
  */
120
121
  export declare function compactionContextTokens(providerContextTokens: number, storedConversationEstimate: number): number;
121
122
  export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings): number;
122
- /**
123
- * Estimate token count for a message using cl100k_base via the native
124
- * tokenizer. This is not Claude's first-party tokenizer (Anthropic doesn't
125
- * publish one) but is within ~5–10% across English/code text.
126
- *
127
- * `excludeEncryptedReasoning` drops opaque provider reasoning payloads
128
- * (`thinkingSignature`, `redactedThinking`) from the estimate. Those are billed
129
- * by the provider on replay, so the default counts them — but their *local*
130
- * byte size can diverge wildly from what the provider charges, so the
131
- * compaction floor (which only needs the reliably-countable, on-wire-compressible
132
- * content) excludes them to avoid false triggers on thinking-heavy turns.
133
- */
134
- export declare function estimateTokens(message: AgentMessage, options?: {
135
- excludeEncryptedReasoning?: boolean;
136
- }): number;
137
123
  /**
138
124
  * Find the user message (or bashExecution) that starts the turn containing the given entry index.
139
125
  * Returns -1 if no turn start found before the index.
@@ -164,7 +150,7 @@ export interface CutPointResult {
164
150
  *
165
151
  * Only considers entries between `startIndex` and `endIndex` (exclusive).
166
152
  */
167
- export declare function findCutPoint(entries: SessionEntry[], startIndex: number, endIndex: number, keepRecentTokens: number): CutPointResult;
153
+ export declare function findCutPoint(entries: SessionEntry[], tokenizer: Tokenizer, startIndex: number, endIndex: number, keepRecentTokens: number): CutPointResult;
168
154
  export declare const AUTO_HANDOFF_THRESHOLD_FOCUS: string;
169
155
  /**
170
156
  * Generate a summary of the conversation using the LLM.
@@ -310,6 +296,22 @@ export interface CompactionPreparation {
310
296
  /** Compaction settions from settings.jsonl */
311
297
  settings: CompactionSettings;
312
298
  }
299
+ /**
300
+ * Whether a prior remote compaction's provider-native replay can still be read
301
+ * by the active model — the model that assembles the request context on every
302
+ * turn. A local compaction (no remote preserve) always can: it holds a real
303
+ * textual summary. A remote compaction (V2 or V1) only can when the active model
304
+ * shares the blob's provider AND remote replay is still enabled; otherwise the
305
+ * active model's encoder drops the payload (see `getOpenAIResponsesHistoryPayload`)
306
+ * and only the opaque placeholder summary survives, so the caller must re-expand
307
+ * the originals into a portable local summary rather than strand that history.
308
+ *
309
+ * Judged against the ACTIVE model, not the compaction candidate set: a role
310
+ * model (e.g. `modelRoles.smol`) that still maps to the blob's provider does not
311
+ * let the active model replay it, so keying reuse on "any candidate shares the
312
+ * provider" left a provider-switched session permanently context-less (#6343).
313
+ */
314
+ export declare function remotePreserveReusable(preserveData: Record<string, unknown> | undefined, activeModel: Model, settings: CompactionSettings): boolean;
313
315
  /**
314
316
  * Index of the newest compaction entry the active model can actually read, or
315
317
  * `-1` when none can.
@@ -323,7 +325,12 @@ export interface CompactionPreparation {
323
325
  * entries that no summary covers.
324
326
  */
325
327
  export declare function findReadableCompactionIndex(pathEntries: SessionEntry[], settings: CompactionSettings, activeModel?: Model): number;
326
- export declare function prepareCompaction(pathEntries: SessionEntry[], settings: CompactionSettings, activeModel?: Model): CompactionPreparation | undefined;
328
+ /**
329
+ * Pass the caller's warm `tokenizer` (the Agent's for the active model) so the
330
+ * full-branch estimate walk hits its memo; the cold default is for one-shot
331
+ * callers that have no live agent.
332
+ */
333
+ export declare function prepareCompaction(pathEntries: SessionEntry[], settings: CompactionSettings, activeModel?: Model, tokenizer?: Tokenizer): CompactionPreparation | undefined;
327
334
  /**
328
335
  * Generate summaries for compaction using prepared data.
329
336
  * Returns CompactionResult - SessionManager adds id/parentId when saving.
@@ -10,4 +10,5 @@ export * from "./messages.js";
10
10
  export * from "./openai.js";
11
11
  export * from "./pruning.js";
12
12
  export * from "./shake.js";
13
+ export * from "./transcript-tokens.js";
13
14
  export * from "./utils.js";
@@ -5,16 +5,18 @@ import type { AgentMessage } from "../types.js";
5
5
  * unregister function. The coding-agent `convertToLlm` memo registers here.
6
6
  */
7
7
  export declare function registerMessageCacheInvalidator(invalidate: (message: AgentMessage) => void): () => void;
8
+ /**
9
+ * Current estimate version of `message` (0 until first invalidation). A
10
+ * `Tokenizer` memo entry stamped with an older version is stale and must be
11
+ * recounted.
12
+ */
13
+ export declare function messageEstimateVersion(message: AgentMessage): number;
8
14
  /**
9
15
  * True when this message's estimate is safe to cache by identity. Non-assistants
10
16
  * are immutable once appended; assistants are cached only once settled (see the
11
17
  * settle-gate invariant above).
12
18
  */
13
19
  export declare function isEstimateCacheable(message: AgentMessage): boolean;
14
- /** Read a cached estimate for the given option split, or `undefined` on miss. */
15
- export declare function readEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean): number | undefined;
16
- /** Store an estimate for the given option split. */
17
- export declare function writeEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean, value: number): void;
18
20
  /**
19
21
  * Drop every cached derivation of `message` after an in-place rewrite. Owners of
20
22
  * mutation (prune, shake, strip-images) call this at the mutation seam so the
@@ -32,6 +32,10 @@ export interface CompactionSummaryMessage {
32
32
  summary: string;
33
33
  shortSummary?: string;
34
34
  tokensBefore: number;
35
+ /** Estimated context tokens after the rewrite (display metadata). */
36
+ tokensAfter?: number;
37
+ /** Harness compaction method that produced this summary (display metadata). */
38
+ method?: string;
35
39
  providerPayload?: ProviderPayload;
36
40
  /** Runtime-only ordered archive blocks for snapcompact: old text region,
37
41
  * imaged middle, then new text region. When present, `summary` is already
@@ -56,7 +60,19 @@ export type ConvertToLlm = (messages: AgentMessage[]) => Message[];
56
60
  export declare function renderBranchSummaryContext(summary: string): string;
57
61
  export declare function renderCompactionSummaryContext(summary: string): string;
58
62
  export declare function createBranchSummaryMessage(summary: string, fromId: string, timestamp: string): BranchSummaryMessage;
59
- export declare function createCompactionSummaryMessage(summary: string, tokensBefore: number, timestamp: string, shortSummary?: string, providerPayload?: ProviderPayload, images?: ImageContent[], blocks?: (TextContent | ImageContent)[], warning?: string): CompactionSummaryMessage;
63
+ /** Optional metadata for {@link createCompactionSummaryMessage}. */
64
+ export interface CompactionSummaryMessageOptions {
65
+ shortSummary?: string;
66
+ providerPayload?: ProviderPayload;
67
+ images?: ImageContent[];
68
+ blocks?: (TextContent | ImageContent)[];
69
+ warning?: string;
70
+ /** Harness compaction method that produced this summary (e.g. "remote", "soft", "handoff"). */
71
+ method?: string;
72
+ /** Estimated context tokens after the rewrite, for display alongside `tokensBefore`. */
73
+ tokensAfter?: number;
74
+ }
75
+ export declare function createCompactionSummaryMessage(summary: string, tokensBefore: number, timestamp: string, options?: CompactionSummaryMessageOptions): CompactionSummaryMessage;
60
76
  export declare function createCustomMessage(customType: string, content: string | (TextContent | ImageContent)[], display: boolean, details: unknown | undefined, timestamp: string, attribution?: MessageAttribution): CustomMessage;
61
77
  /**
62
78
  * Transform a single core-domain agent message to its LLM form; `undefined`
@@ -15,6 +15,7 @@
15
15
  * with `{ summary, shortSummary? }`.
16
16
  */
17
17
  import type { CodexCompactionContext, FetchImpl, Message, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
18
+ import { Tokenizer } from "../tokenizer.js";
18
19
  export * from "./compaction-v2-streaming.js";
19
20
  export declare const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
20
21
  /**
@@ -39,7 +40,7 @@ export interface TrimRemoteCompactionInputResult {
39
40
  * outputs keeps call/result pairing and all earlier assistant/reasoning history,
40
41
  * matching Codex's recovery path for oversized tool turns.
41
42
  */
42
- export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
43
+ export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, tokenizer: Tokenizer, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
43
44
  export type OpenAiRemoteCompactionItem = {
44
45
  type: "compaction" | "compaction_summary";
45
46
  encrypted_content?: string;
@@ -1,6 +1,7 @@
1
1
  /**
2
2
  * Tool output pruning utilities for compaction.
3
3
  */
4
+ import type { Tokenizer } from "../tokenizer.js";
4
5
  import type { SessionEntry } from "./entries.js";
5
6
  import { type ProtectedToolMatcher } from "./tool-protection.js";
6
7
  export interface PruneConfig {
@@ -90,8 +91,8 @@ export interface SupersedePruneConfig {
90
91
  * the provider cache is cold anyway (then all still-sent candidates flush).
91
92
  * Never mutates entries before `keepBoundaryId` (summarized away — not sent).
92
93
  */
93
- export declare function pruneSupersededToolResults(entries: SessionEntry[], config: SupersedePruneConfig): PruneResult;
94
- export declare function pruneToolOutputs(entries: SessionEntry[], config?: PruneConfig): PruneResult;
94
+ export declare function pruneSupersededToolResults(entries: SessionEntry[], tokenizer: Tokenizer, config: SupersedePruneConfig): PruneResult;
95
+ export declare function pruneToolOutputs(entries: SessionEntry[], tokenizer: Tokenizer, config?: PruneConfig): PruneResult;
95
96
  /**
96
97
  * Supersede key for the `read` tool: the file path with the trailing line/raw
97
98
  * selector stripped (the read tool's own splitter grammar via
@@ -9,6 +9,7 @@
9
9
  *
10
10
  * Layering mirrors `pruning.ts`: no I/O here.
11
11
  */
12
+ import type { Tokenizer } from "../tokenizer.js";
12
13
  import type { CustomMessageEntry, SessionEntry, SessionMessageEntry } from "./entries.js";
13
14
  import { type ProtectedToolMatcher } from "./tool-protection.js";
14
15
  export interface ShakeConfig {
@@ -77,7 +78,7 @@ export type ShakeRegion = ToolResultShakeRegion | BlockShakeRegion;
77
78
  * and regions never span a message boundary. When the combined estimated
78
79
  * savings is below `minSavings`, returns `[]` (no-op).
79
80
  */
80
- export declare function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig): ShakeRegion[];
81
+ export declare function collectShakeRegions(entries: SessionEntry[], tokenizer: Tokenizer, config: ShakeConfig): ShakeRegion[];
81
82
  /**
82
83
  * Pure mutation: replace a single region's content in place.
83
84
  *
@@ -0,0 +1,76 @@
1
+ /**
2
+ * Provider-anchored transcript token accounting.
3
+ *
4
+ * Local tokenization is the expensive way to answer "how big is this
5
+ * conversation?" — and usually the wrong one, because the provider already
6
+ * answered it. Every settled assistant turn carries `usage` covering the exact
7
+ * prompt it was sent: the system prompt, the tool schemas, and every message up
8
+ * to and including itself. The only genuinely unaccounted-for text is the tail
9
+ * appended *after* that turn.
10
+ *
11
+ * These helpers locate the newest trustworthy usage report and tokenize only
12
+ * that tail, so a long session pays counting proportional to one turn instead
13
+ * of to the whole transcript, every turn.
14
+ *
15
+ * Trust rules for an anchor (mirroring the provider contract):
16
+ * - Assistant role only — nothing else carries `usage`.
17
+ * - Not `aborted` / `error`: those turns report partial or zero usage.
18
+ * - `hasContextTokenUsage(usage)`: the report must carry usable context numbers.
19
+ */
20
+ import type { AssistantMessage } from "@oh-my-pi/pi-ai";
21
+ import type { Tokenizer } from "../tokenizer.js";
22
+ import type { AgentMessage } from "../types.js";
23
+ /** A provider usage report that accounts for a prefix of the transcript. */
24
+ export interface TranscriptUsageAnchor {
25
+ /** Index in the scanned array; messages at or before it are provider-accounted. */
26
+ index: number;
27
+ /** The anchoring assistant turn. */
28
+ message: AssistantMessage;
29
+ /** Conversation tokens the provider reported for that prompt. */
30
+ tokens: number;
31
+ }
32
+ /**
33
+ * Whether this message's provider usage may anchor transcript accounting.
34
+ *
35
+ * The single home for the trust rules — every anchor scan MUST route through
36
+ * it so a stale-usage rule can never drift between the transcript walkers and
37
+ * the session-entry walkers.
38
+ */
39
+ export declare function isTranscriptUsageAnchor(message: AgentMessage): message is AssistantMessage;
40
+ /**
41
+ * Newest assistant turn in `messages[fromIndex..]` whose usage can anchor the
42
+ * transcript, or `undefined` when none qualifies (fresh context, or every
43
+ * recent turn aborted/errored).
44
+ *
45
+ * `fromIndex` excludes turns whose usage is stale — anything a compaction
46
+ * summarized away describes a prompt that is no longer sent.
47
+ */
48
+ export declare function findTranscriptUsageAnchor(messages: readonly AgentMessage[], fromIndex?: number): TranscriptUsageAnchor | undefined;
49
+ /** Options for {@link estimateTranscriptTokens}. */
50
+ export interface TranscriptTokenOptions {
51
+ /**
52
+ * Gates the anchor search only: usage at or before this index is stale (a
53
+ * compaction rewrote the prompt it describes) and must not anchor. Content
54
+ * accounting is governed separately by {@link countFromIndex}.
55
+ */
56
+ anchorFromIndex?: number;
57
+ /**
58
+ * First message whose content is counted locally when no anchor is found.
59
+ * Defaults to 0 (count the whole transcript), which is what a floor
60
+ * estimate wants; pass the compaction boundary to skip summarized-away
61
+ * messages entirely.
62
+ */
63
+ countFromIndex?: number;
64
+ /** Forwarded to {@link Tokenizer.countMessage} for every locally counted message. */
65
+ excludeEncryptedReasoning?: boolean;
66
+ }
67
+ /**
68
+ * Conversation tokens for `messages`: the provider's own report for everything
69
+ * it already covers, plus a local count of only the unaccounted-for tail.
70
+ *
71
+ * An anchored result already includes the non-message prefix (system prompt +
72
+ * tool schemas) because the provider charged it; an unanchored result is a
73
+ * message-only sum. Callers that add non-message tokens on top MUST branch on
74
+ * {@link findTranscriptUsageAnchor} rather than assuming one shape.
75
+ */
76
+ export declare function estimateTranscriptTokens(messages: readonly AgentMessage[], tokenizer: Tokenizer, options?: TranscriptTokenOptions): number;
@@ -1,2 +1,78 @@
1
- export declare function countTokens(text: string | string[]): number;
2
- export declare function countTokensConservatively(text: string | string[]): number;
1
+ import type { Model } from "@oh-my-pi/pi-ai";
2
+ import { Encoding } from "@oh-my-pi/pi-natives";
3
+ import type { AgentMessage } from "./types.js";
4
+ /** Maps the catalog-resolved tokenizer family to its native implementation. */
5
+ export declare function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): Encoding | null;
6
+ /**
7
+ * `strict` always pays for an exact native count (the catalog-resolved
8
+ * tokenizer when known, o200k_base otherwise). `approximate` and
9
+ * `upperbound` prefer the same exact count for known tokenizer families or
10
+ * when `PI_TOKENIZER_ACCURATE=1` is set; otherwise they use a cheap heuristic:
11
+ * `approximate` a bytes/4 guess, `upperbound` the raw byte length (never
12
+ * undercounts).
13
+ */
14
+ export type TokenCountMode = "strict" | "approximate" | "upperbound";
15
+ /** Options for {@link Tokenizer.countMessage} / {@link Tokenizer.countMessages}. */
16
+ export interface MessageCountOptions {
17
+ /**
18
+ * Drop opaque provider reasoning payloads (`thinkingSignature`,
19
+ * `redactedThinking`, native server-tool blocks) from the estimate. Those
20
+ * are billed by the provider on replay, so the default counts them — but
21
+ * their *local* byte size can diverge wildly from what the provider
22
+ * charges, so the compaction floor (which only needs the reliably-countable,
23
+ * on-wire-compressible content) excludes them to avoid false triggers on
24
+ * thinking-heavy turns.
25
+ */
26
+ excludeEncryptedReasoning?: boolean;
27
+ }
28
+ /** Verdict from {@link Tokenizer.checkTokenBudget}. */
29
+ export interface TokenBudgetCheck {
30
+ /** Whether the text fits the budget. */
31
+ fits: boolean;
32
+ /**
33
+ * Token count behind the verdict: the exact native count when `exact` is
34
+ * set, otherwise the cheap byte upper bound (which already fit, so it is
35
+ * only an over-estimate of a count known to be under budget).
36
+ */
37
+ tokens: number;
38
+ /** Whether the exact tokenizer had to run because the cheap bound busted. */
39
+ exact: boolean;
40
+ }
41
+ /**
42
+ * Model-aware local token counter. Immutable: the catalog-resolved encoding
43
+ * is fixed at construction, so a cached count can never straddle two
44
+ * encodings. An `Agent` owns one for its active model (swapping the instance
45
+ * when the model's encoding changes); one-shot flows construct their own for
46
+ * the model that will be billed. Known tokenizer families use exact native
47
+ * counts; unknown models keep the fast byte estimate (or o200k when
48
+ * `PI_TOKENIZER_ACCURATE=1`).
49
+ */
50
+ export declare class Tokenizer {
51
+ #private;
52
+ constructor(model?: Pick<Model, "tokenizer"> | null);
53
+ get encoding(): Encoding | null;
54
+ countTokens(text: string | string[], mode?: TokenCountMode): number;
55
+ /**
56
+ * Cheap-first budget probe — the way to ask "does this fit in `budget`
57
+ * tokens?" without tokenizing the world.
58
+ *
59
+ * Byte length is a hard upper bound on token count (every token consumes at
60
+ * least one input byte), so text whose raw bytes already fit the budget
61
+ * cannot possibly exceed it — that verdict is returned without tokenizing at
62
+ * all. Only text that busts the bound is ambiguous, and only that case pays
63
+ * for the exact count. Since the bound overshoots ~4x on ordinary prose, the
64
+ * common "comfortably under budget" answer is free.
65
+ */
66
+ checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck;
67
+ /**
68
+ * Token estimate for one message under this tokenizer's encoding.
69
+ *
70
+ * Settled historical messages are counted once and reused until an owner
71
+ * (prune/shake/strip-images) calls `invalidateMessageCache`; streaming
72
+ * assistants bypass the memo entirely (see the message-cache settle-gate
73
+ * invariant). Image blocks charge a fixed per-image estimate.
74
+ */
75
+ countMessage(message: AgentMessage, options?: MessageCountOptions): number;
76
+ /** Sum of {@link countMessage} over `messages`. */
77
+ countMessages(messages: readonly AgentMessage[], options?: MessageCountOptions): number;
78
+ }
@@ -162,7 +162,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
162
162
  * @example
163
163
  * ```typescript
164
164
  * transformContext: async (messages) => {
165
- * if (estimateTokens(messages) > MAX_TOKENS) {
165
+ * if (agent.tokenizer.countMessages(messages) > MAX_TOKENS) {
166
166
  * return pruneOldMessages(messages);
167
167
  * }
168
168
  * return messages;
package/package.json CHANGED
@@ -1,10 +1,10 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-agent-core",
4
- "version": "17.3.8",
4
+ "version": "17.4.1",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://omp.sh",
7
- "author": "Can Boluk",
7
+ "author": "Stencil Labs, Inc.",
8
8
  "contributors": [
9
9
  "Mario Zechner"
10
10
  ],
@@ -35,16 +35,16 @@
35
35
  "fmt": "biome format --write ."
36
36
  },
37
37
  "dependencies": {
38
- "@oh-my-pi/pi-ai": "17.3.8",
39
- "@oh-my-pi/pi-catalog": "17.3.8",
40
- "@oh-my-pi/pi-natives": "17.3.8",
41
- "@oh-my-pi/pi-utils": "17.3.8",
42
- "@oh-my-pi/pi-wire": "17.3.8",
43
- "@oh-my-pi/snapcompact": "17.3.8",
38
+ "@oh-my-pi/pi-ai": "17.4.1",
39
+ "@oh-my-pi/pi-catalog": "17.4.1",
40
+ "@oh-my-pi/pi-natives": "17.4.1",
41
+ "@oh-my-pi/pi-utils": "17.4.1",
42
+ "@oh-my-pi/pi-wire": "17.4.1",
43
+ "@oh-my-pi/snapcompact": "17.4.1",
44
44
  "@opentelemetry/api": "^1.9.1"
45
45
  },
46
46
  "devDependencies": {
47
- "@oh-my-pi/omptype": "17.3.8",
47
+ "@oh-my-pi/omptype": "17.4.1",
48
48
  "@opentelemetry/context-async-hooks": "^2.9.0",
49
49
  "@opentelemetry/sdk-trace-base": "^2.9.0",
50
50
  "@types/bun": "^1.3.14"
@@ -56,6 +56,8 @@
56
56
  "src",
57
57
  "README.md",
58
58
  "CHANGELOG.md",
59
+ "LICENSE",
60
+ "THIRD-PARTY-NOTICES.txt",
59
61
  "dist/types"
60
62
  ],
61
63
  "exports": {
package/src/agent.ts CHANGED
@@ -37,6 +37,7 @@ import {
37
37
  } from "./agent-loop";
38
38
  import type { AppendOnlyContextManager } from "./append-only-context";
39
39
  import { isProviderRefusalMessage } from "./replay-policy";
40
+ import { Tokenizer, tokenizerEncodingForModel } from "./tokenizer";
40
41
  import type {
41
42
  AgentBeforeModelCall,
42
43
  AgentContext,
@@ -363,7 +364,7 @@ export class Agent {
363
364
  pendingToolCalls: new Set<string>(),
364
365
  error: undefined,
365
366
  };
366
-
367
+ #tokenizer = new Tokenizer(this.#state.model);
367
368
  #listeners = new Set<(e: AgentEvent) => void>();
368
369
  #abortController?: AbortController;
369
370
  #convertToLlm: (messages: AgentMessage[]) => Message[] | Promise<Message[]>;
@@ -459,6 +460,7 @@ export class Agent {
459
460
  if (opts.initialState?.messages) this.#state.messages = opts.initialState.messages.slice();
460
461
  if (opts.initialState?.pendingToolCalls)
461
462
  this.#state.pendingToolCalls = new Set(opts.initialState.pendingToolCalls);
463
+ this.#syncTokenizer(this.#state.model);
462
464
  this.#convertToLlm = opts.convertToLlm || defaultConvertToLlm;
463
465
  this.#transformContext = opts.transformContext;
464
466
  this.#steeringMode = opts.steeringMode || "one-at-a-time";
@@ -720,11 +722,29 @@ export class Agent {
720
722
  set maxRetryDelayMs(value: number | undefined) {
721
723
  this.#maxRetryDelayMs = value;
722
724
  }
723
-
724
725
  get state(): AgentState {
725
726
  return this.#state;
726
727
  }
727
728
 
729
+ /**
730
+ * Tokenizer for the active model. The instance is replaced whenever the
731
+ * active model's encoding changes (see {@link setModel}), so callers must
732
+ * not cache it across model switches.
733
+ */
734
+ get tokenizer(): Tokenizer {
735
+ return this.#tokenizer;
736
+ }
737
+
738
+ /**
739
+ * Swap the tokenizer only when the encoding actually changes, so the warm
740
+ * per-message memo survives same-encoding model switches.
741
+ */
742
+ #syncTokenizer(model: Model | null | undefined): void {
743
+ if (tokenizerEncodingForModel(model) !== this.#tokenizer.encoding) {
744
+ this.#tokenizer = new Tokenizer(model);
745
+ }
746
+ }
747
+
728
748
  get appendOnlyContext(): AppendOnlyContextManager | undefined {
729
749
  return this.#appendOnlyContext;
730
750
  }
@@ -902,8 +922,9 @@ export class Agent {
902
922
  this.#state.systemPrompt = typeof v === "string" ? [v] : v;
903
923
  }
904
924
 
905
- setModel(m: Model) {
906
- this.#state.model = m;
925
+ setModel(model: Model) {
926
+ this.#state.model = model;
927
+ this.#syncTokenizer(model);
907
928
  }
908
929
 
909
930
  setThinkingLevel(l: Effort | undefined) {
@@ -9,8 +9,8 @@ import type { Api, ApiKey, AssistantMessage, Context, Model, SimpleStreamOptions
9
9
  import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
10
10
  import { prompt } from "@oh-my-pi/pi-utils";
11
11
  import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
12
+ import { Tokenizer } from "../tokenizer";
12
13
  import type { AgentMessage } from "../types";
13
- import { estimateTokens } from "./compaction";
14
14
  import type { ReadonlySessionManager, SessionEntry } from "./entries";
15
15
  import {
16
16
  type ConvertToLlm,
@@ -191,7 +191,9 @@ function getMessageFromEntry(entry: SessionEntry): AgentMessage | undefined {
191
191
  return createBranchSummaryMessage(entry.summary, entry.fromId, entry.timestamp);
192
192
 
193
193
  case "compaction":
194
- return createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp, entry.shortSummary);
194
+ return createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp, {
195
+ shortSummary: entry.shortSummary,
196
+ });
195
197
 
196
198
  // These don't contribute to conversation content
197
199
  case "thinking_level_change":
@@ -206,14 +208,14 @@ function getMessageFromEntry(entry: SessionEntry): AgentMessage | undefined {
206
208
  }
207
209
  }
208
210
 
209
- function estimateBranchSummaryTokens(message: AgentMessage): number {
210
- if (message.role !== "toolResult") return estimateTokens(message);
211
+ function estimateBranchSummaryTokens(message: AgentMessage, tokenizer: Tokenizer): number {
212
+ if (message.role !== "toolResult") return tokenizer.countMessage(message);
211
213
  const text = message.content
212
214
  .filter((c): c is { type: "text"; text: string } => c.type === "text")
213
215
  .map(c => c.text)
214
216
  .join("");
215
217
  if (!text) return 0;
216
- return estimateTokens({
218
+ return tokenizer.countMessage({
217
219
  ...message,
218
220
  content: [{ type: "text", text: truncateToolResultForSummary(text) }],
219
221
  });
@@ -232,7 +234,11 @@ function estimateBranchSummaryTokens(message: AgentMessage): number {
232
234
  * @param entries - Entries in chronological order
233
235
  * @param tokenBudget - Maximum tokens to include (0 = no limit)
234
236
  */
235
- export function prepareBranchEntries(entries: SessionEntry[], tokenBudget: number = 0): BranchPreparation {
237
+ export function prepareBranchEntries(
238
+ entries: SessionEntry[],
239
+ tokenizer: Tokenizer,
240
+ tokenBudget: number = 0,
241
+ ): BranchPreparation {
236
242
  const messages: AgentMessage[] = [];
237
243
  const fileOps = createFileOps();
238
244
  let totalTokens = 0;
@@ -264,7 +270,7 @@ export function prepareBranchEntries(entries: SessionEntry[], tokenBudget: numbe
264
270
  // Extract file ops from assistant messages (tool calls)
265
271
  extractFileOpsFromMessage(message, fileOps);
266
272
 
267
- const tokens = estimateBranchSummaryTokens(message);
273
+ const tokens = estimateBranchSummaryTokens(message, tokenizer);
268
274
 
269
275
  // Check budget before adding
270
276
  if (tokenBudget > 0 && totalTokens + tokens > tokenBudget) {
@@ -309,8 +315,9 @@ export async function generateBranchSummary(
309
315
  // Token budget = context window minus reserved space for prompt + response
310
316
  const contextWindow = model.contextWindow || 128000;
311
317
  const tokenBudget = contextWindow - reserveTokens;
318
+ const tokenizer = new Tokenizer(model);
312
319
 
313
- const { messages, fileOps } = prepareBranchEntries(entries, tokenBudget);
320
+ const { messages, fileOps } = prepareBranchEntries(entries, tokenizer, tokenBudget);
314
321
 
315
322
  if (messages.length === 0) {
316
323
  return { summary: "No content to summarize" };
@@ -25,6 +25,7 @@ import {
25
25
  } from "@oh-my-pi/pi-ai/providers/openai-shared";
26
26
  import { captureOpenAIHttpError } from "@oh-my-pi/pi-ai/utils/openai-http";
27
27
  import {
28
+ applyCodexResidencyHeader,
28
29
  CODEX_BASE_URL,
29
30
  getCodexAccountId,
30
31
  OPENAI_HEADER_VALUES,
@@ -416,6 +417,7 @@ function buildCompactionV2Headers(
416
417
  if (accountId) {
417
418
  headers[OPENAI_HEADERS.ACCOUNT_ID] = accountId;
418
419
  }
420
+ applyCodexResidencyHeader(headers, apiKey);
419
421
  if (routingSessionId) {
420
422
  headers[OPENAI_HEADERS.CONVERSATION_ID] = routingSessionId;
421
423
  headers[OPENAI_HEADERS.SESSION_ID] = routingSessionId;