@oh-my-pi/pi-agent-core 17.3.7 → 17.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,33 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [17.4.0] - 2026-08-20
6
+
7
+ ### Breaking Changes
8
+
9
+ - Replaced global token counting functions (`countTokens`, `countTokensConservatively`, `setTokenizerModel`, and `estimateTokens`) with model-scoped, immutable `Tokenizer` instances (`agent.tokenizer`). Use `tokenizer.countTokens(text, mode?)`, `tokenizer.countMessage(message)`, or `tokenizer.countMessages(messages)`.
10
+ - Updated context management functions (`findCutPoint`, `prepareBranchEntries`, `collectShakeRegions`, `pruneToolOutputs`, `pruneSupersededToolResults`, and `trimRemoteCompactionInputToContextWindow`) to require an explicit `Tokenizer` instance.
11
+
12
+ ### Added
13
+
14
+ - Added `Tokenizer.checkTokenBudget(text, budget)` to efficiently verify if text fits within a token limit using fast byte-bound checks before falling back to full tokenization.
15
+ - Added provider-anchored transcript token estimation (`findTranscriptUsageAnchor`, `isTranscriptUsageAnchor`, `estimateTranscriptTokens`) to calculate transcript token counts incrementally from the latest reported assistant turn usage.
16
+ - Added `remotePreserveReusable()` to check whether a previous remote compaction payload remains reusable with the active model.
17
+
18
+ ### Changed
19
+
20
+ - Expanded native tokenizer support across catalog models, adding exact embedded token counting for Claude, Qwen 3.5+, DeepSeek V3/V4/R1, Kimi K2/K3, and GLM-5+ models. `Tokenizer` now constructs from a resolved catalog `Model`.
21
+ - `createCompactionSummaryMessage` takes an options object after `(summary, tokensBefore, timestamp)`; `CompactionSummaryMessage` gained optional `method` and `tokensAfter` display metadata.
22
+
23
+ ## [17.3.8] - 2026-08-19
24
+
25
+ ### Fixed
26
+
27
+ - Fixed `/compact` (and automatic compaction) resurrecting pre-`/clear` conversation turns: `prepareCompaction` now honors the latest `reset_boundary`, so a compaction after an in-place `/clear` only summarizes messages created after the reset ([#8718](https://github.com/can1357/oh-my-pi/issues/8718)).
28
+ - Hardened compaction summarization against prompt injection: conversation history and previous summaries are now treated as untrusted, and embedded `<conversation>`/`<previous-summary>` boundary tags are neutralized before prompt assembly ([#8727](https://github.com/can1357/oh-my-pi/pull/8727) by [@koopmannleon19977-cmyk](https://github.com/koopmannleon19977-cmyk)).
29
+ - Compaction summarization input is now bounded to the summary model's context (windowed fold for oversized spans) and deterministic context-overflow 400s are no longer retried up to the full retry budget; artifact ids containing `503` no longer misclassify hard 400s as transient.
30
+ - Fixed remote compaction mirroring the #8789 Responses shape: `buildOpenAiNativeHistory` now hoists an assistant `message` wedged between a tool-call batch and its outputs ahead of the batch, so compaction requests to strict opencode-go gateways match the canonical `message(s) → calls → outputs` order ([#8789](https://github.com/can1357/oh-my-pi/issues/8789)).
31
+
5
32
  ## [17.3.5] - 2026-08-16
6
33
 
7
34
  ### Added
@@ -2,6 +2,7 @@ import { type ApiKey, type AssistantMessage, type AssistantMessageEvent, type Co
2
2
  import type { Dialect } from "@oh-my-pi/pi-ai/dialect";
3
3
  import type { HarmonyAuditEvent } from "@oh-my-pi/pi-ai/utils/harmony-leak";
4
4
  import type { AppendOnlyContextManager } from "./append-only-context.js";
5
+ import { Tokenizer } from "./tokenizer.js";
5
6
  import type { AgentBeforeModelCall, AgentEvent, AgentLoopConfig, AgentMessage, AgentState, AgentTool, AgentToolContext, AgentTurnEndContext, AsideMessage, StreamFn, ToolCallContext, ToolChoiceDirective } from "./types.js";
6
7
  export declare class AgentBusyError extends Error {
7
8
  constructor(message?: string);
@@ -354,6 +355,12 @@ export declare class Agent {
354
355
  */
355
356
  set maxRetryDelayMs(value: number | undefined);
356
357
  get state(): AgentState;
358
+ /**
359
+ * Tokenizer for the active model. The instance is replaced whenever the
360
+ * active model's encoding changes (see {@link setModel}), so callers must
361
+ * not cache it across model switches.
362
+ */
363
+ get tokenizer(): Tokenizer;
357
364
  get appendOnlyContext(): AppendOnlyContextManager | undefined;
358
365
  setAppendOnlyContext(manager?: AppendOnlyContextManager): void;
359
366
  /**
@@ -401,7 +408,7 @@ export declare class Agent {
401
408
  setAsideMessageProvider(fn: (() => AsideMessage[] | Promise<AsideMessage[]>) | undefined): void;
402
409
  emitExternalEvent(event: AgentEvent): void;
403
410
  setSystemPrompt(v: string[] | string): void;
404
- setModel(m: Model): void;
411
+ setModel(model: Model): void;
405
412
  setThinkingLevel(l: Effort | undefined): void;
406
413
  setDisableReasoning(disabled: boolean): void;
407
414
  setSteeringMode(mode: "all" | "one-at-a-time"): void;
@@ -6,6 +6,7 @@
6
6
  */
7
7
  import type { Api, ApiKey, AssistantMessage, Context, Model, SimpleStreamOptions } from "@oh-my-pi/pi-ai";
8
8
  import { type AgentTelemetry } from "../telemetry.js";
9
+ import { Tokenizer } from "../tokenizer.js";
9
10
  import type { AgentMessage } from "../types.js";
10
11
  import type { ReadonlySessionManager, SessionEntry } from "./entries.js";
11
12
  import { type ConvertToLlm } from "./messages.js";
@@ -91,7 +92,7 @@ export declare function collectEntriesForBranchSummary(session: ReadonlySessionM
91
92
  * @param entries - Entries in chronological order
92
93
  * @param tokenBudget - Maximum tokens to include (0 = no limit)
93
94
  */
94
- export declare function prepareBranchEntries(entries: SessionEntry[], tokenBudget?: number): BranchPreparation;
95
+ export declare function prepareBranchEntries(entries: SessionEntry[], tokenizer: Tokenizer, tokenBudget?: number): BranchPreparation;
95
96
  /**
96
97
  * Generate a summary of abandoned branch entries.
97
98
  *
@@ -7,6 +7,7 @@
7
7
  import { type Api, type ApiKey, type AssistantMessage, type CodexCompactionContext, type Context, type FetchImpl, type MessageAttribution, type Model, type OneshotRetryOptions, type ProviderSessionState, type SimpleStreamOptions, type Tool, type Usage } from "@oh-my-pi/pi-ai";
8
8
  import { type AgentTelemetry } from "../telemetry.js";
9
9
  import { ThinkingLevel } from "../thinking.js";
10
+ import { Tokenizer } from "../tokenizer.js";
10
11
  import type { AgentMessage } from "../types.js";
11
12
  import type { SessionEntry } from "./entries.js";
12
13
  import { type ConvertToLlm } from "./messages.js";
@@ -119,21 +120,6 @@ export declare function shouldCompact(contextTokens: number, contextWindow: numb
119
120
  */
120
121
  export declare function compactionContextTokens(providerContextTokens: number, storedConversationEstimate: number): number;
121
122
  export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings): number;
122
- /**
123
- * Estimate token count for a message using cl100k_base via the native
124
- * tokenizer. This is not Claude's first-party tokenizer (Anthropic doesn't
125
- * publish one) but is within ~5–10% across English/code text.
126
- *
127
- * `excludeEncryptedReasoning` drops opaque provider reasoning payloads
128
- * (`thinkingSignature`, `redactedThinking`) from the estimate. Those are billed
129
- * by the provider on replay, so the default counts them — but their *local*
130
- * byte size can diverge wildly from what the provider charges, so the
131
- * compaction floor (which only needs the reliably-countable, on-wire-compressible
132
- * content) excludes them to avoid false triggers on thinking-heavy turns.
133
- */
134
- export declare function estimateTokens(message: AgentMessage, options?: {
135
- excludeEncryptedReasoning?: boolean;
136
- }): number;
137
123
  /**
138
124
  * Find the user message (or bashExecution) that starts the turn containing the given entry index.
139
125
  * Returns -1 if no turn start found before the index.
@@ -164,7 +150,7 @@ export interface CutPointResult {
164
150
  *
165
151
  * Only considers entries between `startIndex` and `endIndex` (exclusive).
166
152
  */
167
- export declare function findCutPoint(entries: SessionEntry[], startIndex: number, endIndex: number, keepRecentTokens: number): CutPointResult;
153
+ export declare function findCutPoint(entries: SessionEntry[], tokenizer: Tokenizer, startIndex: number, endIndex: number, keepRecentTokens: number): CutPointResult;
168
154
  export declare const AUTO_HANDOFF_THRESHOLD_FOCUS: string;
169
155
  /**
170
156
  * Generate a summary of the conversation using the LLM.
@@ -310,7 +296,41 @@ export interface CompactionPreparation {
310
296
  /** Compaction settions from settings.jsonl */
311
297
  settings: CompactionSettings;
312
298
  }
313
- export declare function prepareCompaction(pathEntries: SessionEntry[], settings: CompactionSettings, activeModel?: Model): CompactionPreparation | undefined;
299
+ /**
300
+ * Whether a prior remote compaction's provider-native replay can still be read
301
+ * by the active model — the model that assembles the request context on every
302
+ * turn. A local compaction (no remote preserve) always can: it holds a real
303
+ * textual summary. A remote compaction (V2 or V1) only can when the active model
304
+ * shares the blob's provider AND remote replay is still enabled; otherwise the
305
+ * active model's encoder drops the payload (see `getOpenAIResponsesHistoryPayload`)
306
+ * and only the opaque placeholder summary survives, so the caller must re-expand
307
+ * the originals into a portable local summary rather than strand that history.
308
+ *
309
+ * Judged against the ACTIVE model, not the compaction candidate set: a role
310
+ * model (e.g. `modelRoles.smol`) that still maps to the blob's provider does not
311
+ * let the active model replay it, so keying reuse on "any candidate shares the
312
+ * provider" left a provider-switched session permanently context-less (#6343).
313
+ */
314
+ export declare function remotePreserveReusable(preserveData: Record<string, unknown> | undefined, activeModel: Model, settings: CompactionSettings): boolean;
315
+ /**
316
+ * Index of the newest compaction entry the active model can actually read, or
317
+ * `-1` when none can.
318
+ *
319
+ * A provider-native remote compaction (V2 or V1) stores an opaque replay payload
320
+ * and only a placeholder summary, so for any OTHER provider that entry
321
+ * summarizes nothing and the history behind it is still live context. Callers
322
+ * must therefore treat it as absent: `prepareCompaction` re-expands past it and
323
+ * summarizes those messages locally, and the maintenance ops that use the
324
+ * compaction boundary to skip "already summarized away" entries must not skip
325
+ * entries that no summary covers.
326
+ */
327
+ export declare function findReadableCompactionIndex(pathEntries: SessionEntry[], settings: CompactionSettings, activeModel?: Model): number;
328
+ /**
329
+ * Pass the caller's warm `tokenizer` (the Agent's for the active model) so the
330
+ * full-branch estimate walk hits its memo; the cold default is for one-shot
331
+ * callers that have no live agent.
332
+ */
333
+ export declare function prepareCompaction(pathEntries: SessionEntry[], settings: CompactionSettings, activeModel?: Model, tokenizer?: Tokenizer): CompactionPreparation | undefined;
314
334
  /**
315
335
  * Generate summaries for compaction using prepared data.
316
336
  * Returns CompactionResult - SessionManager adds id/parentId when saving.
@@ -102,9 +102,18 @@ export interface ModeChangeEntry extends SessionEntryBase {
102
102
  /** Optional mode-specific data (e.g. plan file path) */
103
103
  data?: Record<string, unknown>;
104
104
  }
105
+ /**
106
+ * Durable context-reset marker recorded by an in-place `/clear`. It carries no
107
+ * payload — its presence on the branch means every entry before it was dropped
108
+ * from the model context, so context assembly and compaction start after the
109
+ * latest one. The full pre-reset history stays on disk for transcript export.
110
+ */
111
+ export interface ResetBoundaryEntry extends SessionEntryBase {
112
+ type: "reset_boundary";
113
+ }
105
114
  export interface CustomCompactionSessionEntries {
106
115
  }
107
- export type SessionEntry = SessionMessageEntry | ThinkingLevelChangeEntry | ModelChangeEntry | ServiceTierChangeEntry | CompactionEntry | BranchSummaryEntry | CustomEntry | CustomMessageEntry | LabelEntry | TitleChangeEntry | TtsrInjectionEntry | SessionInitEntry | ModeChangeEntry | CustomCompactionSessionEntries[keyof CustomCompactionSessionEntries];
116
+ export type SessionEntry = SessionMessageEntry | ThinkingLevelChangeEntry | ModelChangeEntry | ServiceTierChangeEntry | CompactionEntry | BranchSummaryEntry | CustomEntry | CustomMessageEntry | LabelEntry | TitleChangeEntry | TtsrInjectionEntry | SessionInitEntry | ModeChangeEntry | ResetBoundaryEntry | CustomCompactionSessionEntries[keyof CustomCompactionSessionEntries];
108
117
  export interface ReadonlySessionManager {
109
118
  getBranch(leafId?: string | null): SessionEntry[];
110
119
  getEntry(id: string): SessionEntry | undefined;
@@ -10,4 +10,5 @@ export * from "./messages.js";
10
10
  export * from "./openai.js";
11
11
  export * from "./pruning.js";
12
12
  export * from "./shake.js";
13
+ export * from "./transcript-tokens.js";
13
14
  export * from "./utils.js";
@@ -5,16 +5,18 @@ import type { AgentMessage } from "../types.js";
5
5
  * unregister function. The coding-agent `convertToLlm` memo registers here.
6
6
  */
7
7
  export declare function registerMessageCacheInvalidator(invalidate: (message: AgentMessage) => void): () => void;
8
+ /**
9
+ * Current estimate version of `message` (0 until first invalidation). A
10
+ * `Tokenizer` memo entry stamped with an older version is stale and must be
11
+ * recounted.
12
+ */
13
+ export declare function messageEstimateVersion(message: AgentMessage): number;
8
14
  /**
9
15
  * True when this message's estimate is safe to cache by identity. Non-assistants
10
16
  * are immutable once appended; assistants are cached only once settled (see the
11
17
  * settle-gate invariant above).
12
18
  */
13
19
  export declare function isEstimateCacheable(message: AgentMessage): boolean;
14
- /** Read a cached estimate for the given option split, or `undefined` on miss. */
15
- export declare function readEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean): number | undefined;
16
- /** Store an estimate for the given option split. */
17
- export declare function writeEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean, value: number): void;
18
20
  /**
19
21
  * Drop every cached derivation of `message` after an in-place rewrite. Owners of
20
22
  * mutation (prune, shake, strip-images) call this at the mutation seam so the
@@ -32,6 +32,10 @@ export interface CompactionSummaryMessage {
32
32
  summary: string;
33
33
  shortSummary?: string;
34
34
  tokensBefore: number;
35
+ /** Estimated context tokens after the rewrite (display metadata). */
36
+ tokensAfter?: number;
37
+ /** Harness compaction method that produced this summary (display metadata). */
38
+ method?: string;
35
39
  providerPayload?: ProviderPayload;
36
40
  /** Runtime-only ordered archive blocks for snapcompact: old text region,
37
41
  * imaged middle, then new text region. When present, `summary` is already
@@ -56,7 +60,19 @@ export type ConvertToLlm = (messages: AgentMessage[]) => Message[];
56
60
  export declare function renderBranchSummaryContext(summary: string): string;
57
61
  export declare function renderCompactionSummaryContext(summary: string): string;
58
62
  export declare function createBranchSummaryMessage(summary: string, fromId: string, timestamp: string): BranchSummaryMessage;
59
- export declare function createCompactionSummaryMessage(summary: string, tokensBefore: number, timestamp: string, shortSummary?: string, providerPayload?: ProviderPayload, images?: ImageContent[], blocks?: (TextContent | ImageContent)[], warning?: string): CompactionSummaryMessage;
63
+ /** Optional metadata for {@link createCompactionSummaryMessage}. */
64
+ export interface CompactionSummaryMessageOptions {
65
+ shortSummary?: string;
66
+ providerPayload?: ProviderPayload;
67
+ images?: ImageContent[];
68
+ blocks?: (TextContent | ImageContent)[];
69
+ warning?: string;
70
+ /** Harness compaction method that produced this summary (e.g. "remote", "soft", "handoff"). */
71
+ method?: string;
72
+ /** Estimated context tokens after the rewrite, for display alongside `tokensBefore`. */
73
+ tokensAfter?: number;
74
+ }
75
+ export declare function createCompactionSummaryMessage(summary: string, tokensBefore: number, timestamp: string, options?: CompactionSummaryMessageOptions): CompactionSummaryMessage;
60
76
  export declare function createCustomMessage(customType: string, content: string | (TextContent | ImageContent)[], display: boolean, details: unknown | undefined, timestamp: string, attribution?: MessageAttribution): CustomMessage;
61
77
  /**
62
78
  * Transform a single core-domain agent message to its LLM form; `undefined`
@@ -15,6 +15,7 @@
15
15
  * with `{ summary, shortSummary? }`.
16
16
  */
17
17
  import type { CodexCompactionContext, FetchImpl, Message, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
18
+ import { Tokenizer } from "../tokenizer.js";
18
19
  export * from "./compaction-v2-streaming.js";
19
20
  export declare const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
20
21
  /**
@@ -39,7 +40,7 @@ export interface TrimRemoteCompactionInputResult {
39
40
  * outputs keeps call/result pairing and all earlier assistant/reasoning history,
40
41
  * matching Codex's recovery path for oversized tool turns.
41
42
  */
42
- export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
43
+ export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, tokenizer: Tokenizer, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
43
44
  export type OpenAiRemoteCompactionItem = {
44
45
  type: "compaction" | "compaction_summary";
45
46
  encrypted_content?: string;
@@ -1,6 +1,7 @@
1
1
  /**
2
2
  * Tool output pruning utilities for compaction.
3
3
  */
4
+ import type { Tokenizer } from "../tokenizer.js";
4
5
  import type { SessionEntry } from "./entries.js";
5
6
  import { type ProtectedToolMatcher } from "./tool-protection.js";
6
7
  export interface PruneConfig {
@@ -90,8 +91,8 @@ export interface SupersedePruneConfig {
90
91
  * the provider cache is cold anyway (then all still-sent candidates flush).
91
92
  * Never mutates entries before `keepBoundaryId` (summarized away — not sent).
92
93
  */
93
- export declare function pruneSupersededToolResults(entries: SessionEntry[], config: SupersedePruneConfig): PruneResult;
94
- export declare function pruneToolOutputs(entries: SessionEntry[], config?: PruneConfig): PruneResult;
94
+ export declare function pruneSupersededToolResults(entries: SessionEntry[], tokenizer: Tokenizer, config: SupersedePruneConfig): PruneResult;
95
+ export declare function pruneToolOutputs(entries: SessionEntry[], tokenizer: Tokenizer, config?: PruneConfig): PruneResult;
95
96
  /**
96
97
  * Supersede key for the `read` tool: the file path with the trailing line/raw
97
98
  * selector stripped (the read tool's own splitter grammar via
@@ -9,6 +9,7 @@
9
9
  *
10
10
  * Layering mirrors `pruning.ts`: no I/O here.
11
11
  */
12
+ import type { Tokenizer } from "../tokenizer.js";
12
13
  import type { CustomMessageEntry, SessionEntry, SessionMessageEntry } from "./entries.js";
13
14
  import { type ProtectedToolMatcher } from "./tool-protection.js";
14
15
  export interface ShakeConfig {
@@ -77,7 +78,7 @@ export type ShakeRegion = ToolResultShakeRegion | BlockShakeRegion;
77
78
  * and regions never span a message boundary. When the combined estimated
78
79
  * savings is below `minSavings`, returns `[]` (no-op).
79
80
  */
80
- export declare function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig): ShakeRegion[];
81
+ export declare function collectShakeRegions(entries: SessionEntry[], tokenizer: Tokenizer, config: ShakeConfig): ShakeRegion[];
81
82
  /**
82
83
  * Pure mutation: replace a single region's content in place.
83
84
  *
@@ -0,0 +1,76 @@
1
+ /**
2
+ * Provider-anchored transcript token accounting.
3
+ *
4
+ * Local tokenization is the expensive way to answer "how big is this
5
+ * conversation?" — and usually the wrong one, because the provider already
6
+ * answered it. Every settled assistant turn carries `usage` covering the exact
7
+ * prompt it was sent: the system prompt, the tool schemas, and every message up
8
+ * to and including itself. The only genuinely unaccounted-for text is the tail
9
+ * appended *after* that turn.
10
+ *
11
+ * These helpers locate the newest trustworthy usage report and tokenize only
12
+ * that tail, so a long session pays counting proportional to one turn instead
13
+ * of to the whole transcript, every turn.
14
+ *
15
+ * Trust rules for an anchor (mirroring the provider contract):
16
+ * - Assistant role only — nothing else carries `usage`.
17
+ * - Not `aborted` / `error`: those turns report partial or zero usage.
18
+ * - `hasContextTokenUsage(usage)`: the report must carry usable context numbers.
19
+ */
20
+ import type { AssistantMessage } from "@oh-my-pi/pi-ai";
21
+ import type { Tokenizer } from "../tokenizer.js";
22
+ import type { AgentMessage } from "../types.js";
23
+ /** A provider usage report that accounts for a prefix of the transcript. */
24
+ export interface TranscriptUsageAnchor {
25
+ /** Index in the scanned array; messages at or before it are provider-accounted. */
26
+ index: number;
27
+ /** The anchoring assistant turn. */
28
+ message: AssistantMessage;
29
+ /** Conversation tokens the provider reported for that prompt. */
30
+ tokens: number;
31
+ }
32
+ /**
33
+ * Whether this message's provider usage may anchor transcript accounting.
34
+ *
35
+ * The single home for the trust rules — every anchor scan MUST route through
36
+ * it so a stale-usage rule can never drift between the transcript walkers and
37
+ * the session-entry walkers.
38
+ */
39
+ export declare function isTranscriptUsageAnchor(message: AgentMessage): message is AssistantMessage;
40
+ /**
41
+ * Newest assistant turn in `messages[fromIndex..]` whose usage can anchor the
42
+ * transcript, or `undefined` when none qualifies (fresh context, or every
43
+ * recent turn aborted/errored).
44
+ *
45
+ * `fromIndex` excludes turns whose usage is stale — anything a compaction
46
+ * summarized away describes a prompt that is no longer sent.
47
+ */
48
+ export declare function findTranscriptUsageAnchor(messages: readonly AgentMessage[], fromIndex?: number): TranscriptUsageAnchor | undefined;
49
+ /** Options for {@link estimateTranscriptTokens}. */
50
+ export interface TranscriptTokenOptions {
51
+ /**
52
+ * Gates the anchor search only: usage at or before this index is stale (a
53
+ * compaction rewrote the prompt it describes) and must not anchor. Content
54
+ * accounting is governed separately by {@link countFromIndex}.
55
+ */
56
+ anchorFromIndex?: number;
57
+ /**
58
+ * First message whose content is counted locally when no anchor is found.
59
+ * Defaults to 0 (count the whole transcript), which is what a floor
60
+ * estimate wants; pass the compaction boundary to skip summarized-away
61
+ * messages entirely.
62
+ */
63
+ countFromIndex?: number;
64
+ /** Forwarded to {@link Tokenizer.countMessage} for every locally counted message. */
65
+ excludeEncryptedReasoning?: boolean;
66
+ }
67
+ /**
68
+ * Conversation tokens for `messages`: the provider's own report for everything
69
+ * it already covers, plus a local count of only the unaccounted-for tail.
70
+ *
71
+ * An anchored result already includes the non-message prefix (system prompt +
72
+ * tool schemas) because the provider charged it; an unanchored result is a
73
+ * message-only sum. Callers that add non-message tokens on top MUST branch on
74
+ * {@link findTranscriptUsageAnchor} rather than assuming one shape.
75
+ */
76
+ export declare function estimateTranscriptTokens(messages: readonly AgentMessage[], tokenizer: Tokenizer, options?: TranscriptTokenOptions): number;
@@ -49,6 +49,8 @@ export declare function upsertFileOperations(summary: string, readFiles: string[
49
49
  * Truncate tool results to the same representation used in summarization prompts.
50
50
  */
51
51
  export declare function truncateToolResultForSummary(text: string): string;
52
+ /** Keep untrusted summary input from closing or impersonating harness-owned boundaries. */
53
+ export declare function escapeSummaryBoundaryTags(text: string): string;
52
54
  /**
53
55
  * Serialize LLM messages as plain summary input without provider control tokens.
54
56
  */
@@ -1,2 +1,78 @@
1
- export declare function countTokens(text: string | string[]): number;
2
- export declare function countTokensConservatively(text: string | string[]): number;
1
+ import type { Model } from "@oh-my-pi/pi-ai";
2
+ import { Encoding } from "@oh-my-pi/pi-natives";
3
+ import type { AgentMessage } from "./types.js";
4
+ /** Maps the catalog-resolved tokenizer family to its native implementation. */
5
+ export declare function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): Encoding | null;
6
+ /**
7
+ * `strict` always pays for an exact native count (the catalog-resolved
8
+ * tokenizer when known, o200k_base otherwise). `approximate` and
9
+ * `upperbound` prefer the same exact count for known tokenizer families or
10
+ * when `PI_TOKENIZER_ACCURATE=1` is set; otherwise they use a cheap heuristic:
11
+ * `approximate` a bytes/4 guess, `upperbound` the raw byte length (never
12
+ * undercounts).
13
+ */
14
+ export type TokenCountMode = "strict" | "approximate" | "upperbound";
15
+ /** Options for {@link Tokenizer.countMessage} / {@link Tokenizer.countMessages}. */
16
+ export interface MessageCountOptions {
17
+ /**
18
+ * Drop opaque provider reasoning payloads (`thinkingSignature`,
19
+ * `redactedThinking`, native server-tool blocks) from the estimate. Those
20
+ * are billed by the provider on replay, so the default counts them — but
21
+ * their *local* byte size can diverge wildly from what the provider
22
+ * charges, so the compaction floor (which only needs the reliably-countable,
23
+ * on-wire-compressible content) excludes them to avoid false triggers on
24
+ * thinking-heavy turns.
25
+ */
26
+ excludeEncryptedReasoning?: boolean;
27
+ }
28
+ /** Verdict from {@link Tokenizer.checkTokenBudget}. */
29
+ export interface TokenBudgetCheck {
30
+ /** Whether the text fits the budget. */
31
+ fits: boolean;
32
+ /**
33
+ * Token count behind the verdict: the exact native count when `exact` is
34
+ * set, otherwise the cheap byte upper bound (which already fit, so it is
35
+ * only an over-estimate of a count known to be under budget).
36
+ */
37
+ tokens: number;
38
+ /** Whether the exact tokenizer had to run because the cheap bound busted. */
39
+ exact: boolean;
40
+ }
41
+ /**
42
+ * Model-aware local token counter. Immutable: the catalog-resolved encoding
43
+ * is fixed at construction, so a cached count can never straddle two
44
+ * encodings. An `Agent` owns one for its active model (swapping the instance
45
+ * when the model's encoding changes); one-shot flows construct their own for
46
+ * the model that will be billed. Known tokenizer families use exact native
47
+ * counts; unknown models keep the fast byte estimate (or o200k when
48
+ * `PI_TOKENIZER_ACCURATE=1`).
49
+ */
50
+ export declare class Tokenizer {
51
+ #private;
52
+ constructor(model?: Pick<Model, "tokenizer"> | null);
53
+ get encoding(): Encoding | null;
54
+ countTokens(text: string | string[], mode?: TokenCountMode): number;
55
+ /**
56
+ * Cheap-first budget probe — the way to ask "does this fit in `budget`
57
+ * tokens?" without tokenizing the world.
58
+ *
59
+ * Byte length is a hard upper bound on token count (every token consumes at
60
+ * least one input byte), so text whose raw bytes already fit the budget
61
+ * cannot possibly exceed it — that verdict is returned without tokenizing at
62
+ * all. Only text that busts the bound is ambiguous, and only that case pays
63
+ * for the exact count. Since the bound overshoots ~4x on ordinary prose, the
64
+ * common "comfortably under budget" answer is free.
65
+ */
66
+ checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck;
67
+ /**
68
+ * Token estimate for one message under this tokenizer's encoding.
69
+ *
70
+ * Settled historical messages are counted once and reused until an owner
71
+ * (prune/shake/strip-images) calls `invalidateMessageCache`; streaming
72
+ * assistants bypass the memo entirely (see the message-cache settle-gate
73
+ * invariant). Image blocks charge a fixed per-image estimate.
74
+ */
75
+ countMessage(message: AgentMessage, options?: MessageCountOptions): number;
76
+ /** Sum of {@link countMessage} over `messages`. */
77
+ countMessages(messages: readonly AgentMessage[], options?: MessageCountOptions): number;
78
+ }
@@ -162,7 +162,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
162
162
  * @example
163
163
  * ```typescript
164
164
  * transformContext: async (messages) => {
165
- * if (estimateTokens(messages) > MAX_TOKENS) {
165
+ * if (agent.tokenizer.countMessages(messages) > MAX_TOKENS) {
166
166
  * return pruneOldMessages(messages);
167
167
  * }
168
168
  * return messages;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-agent-core",
4
- "version": "17.3.7",
4
+ "version": "17.4.0",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Can Boluk",
@@ -35,16 +35,16 @@
35
35
  "fmt": "biome format --write ."
36
36
  },
37
37
  "dependencies": {
38
- "@oh-my-pi/pi-ai": "17.3.7",
39
- "@oh-my-pi/pi-catalog": "17.3.7",
40
- "@oh-my-pi/pi-natives": "17.3.7",
41
- "@oh-my-pi/pi-utils": "17.3.7",
42
- "@oh-my-pi/pi-wire": "17.3.7",
43
- "@oh-my-pi/snapcompact": "17.3.7",
38
+ "@oh-my-pi/pi-ai": "17.4.0",
39
+ "@oh-my-pi/pi-catalog": "17.4.0",
40
+ "@oh-my-pi/pi-natives": "17.4.0",
41
+ "@oh-my-pi/pi-utils": "17.4.0",
42
+ "@oh-my-pi/pi-wire": "17.4.0",
43
+ "@oh-my-pi/snapcompact": "17.4.0",
44
44
  "@opentelemetry/api": "^1.9.1"
45
45
  },
46
46
  "devDependencies": {
47
- "@oh-my-pi/omptype": "17.3.7",
47
+ "@oh-my-pi/omptype": "17.4.0",
48
48
  "@opentelemetry/context-async-hooks": "^2.9.0",
49
49
  "@opentelemetry/sdk-trace-base": "^2.9.0",
50
50
  "@types/bun": "^1.3.14"
package/src/agent.ts CHANGED
@@ -37,6 +37,7 @@ import {
37
37
  } from "./agent-loop";
38
38
  import type { AppendOnlyContextManager } from "./append-only-context";
39
39
  import { isProviderRefusalMessage } from "./replay-policy";
40
+ import { Tokenizer, tokenizerEncodingForModel } from "./tokenizer";
40
41
  import type {
41
42
  AgentBeforeModelCall,
42
43
  AgentContext,
@@ -363,7 +364,7 @@ export class Agent {
363
364
  pendingToolCalls: new Set<string>(),
364
365
  error: undefined,
365
366
  };
366
-
367
+ #tokenizer = new Tokenizer(this.#state.model);
367
368
  #listeners = new Set<(e: AgentEvent) => void>();
368
369
  #abortController?: AbortController;
369
370
  #convertToLlm: (messages: AgentMessage[]) => Message[] | Promise<Message[]>;
@@ -459,6 +460,7 @@ export class Agent {
459
460
  if (opts.initialState?.messages) this.#state.messages = opts.initialState.messages.slice();
460
461
  if (opts.initialState?.pendingToolCalls)
461
462
  this.#state.pendingToolCalls = new Set(opts.initialState.pendingToolCalls);
463
+ this.#syncTokenizer(this.#state.model);
462
464
  this.#convertToLlm = opts.convertToLlm || defaultConvertToLlm;
463
465
  this.#transformContext = opts.transformContext;
464
466
  this.#steeringMode = opts.steeringMode || "one-at-a-time";
@@ -720,11 +722,29 @@ export class Agent {
720
722
  set maxRetryDelayMs(value: number | undefined) {
721
723
  this.#maxRetryDelayMs = value;
722
724
  }
723
-
724
725
  get state(): AgentState {
725
726
  return this.#state;
726
727
  }
727
728
 
729
+ /**
730
+ * Tokenizer for the active model. The instance is replaced whenever the
731
+ * active model's encoding changes (see {@link setModel}), so callers must
732
+ * not cache it across model switches.
733
+ */
734
+ get tokenizer(): Tokenizer {
735
+ return this.#tokenizer;
736
+ }
737
+
738
+ /**
739
+ * Swap the tokenizer only when the encoding actually changes, so the warm
740
+ * per-message memo survives same-encoding model switches.
741
+ */
742
+ #syncTokenizer(model: Model | null | undefined): void {
743
+ if (tokenizerEncodingForModel(model) !== this.#tokenizer.encoding) {
744
+ this.#tokenizer = new Tokenizer(model);
745
+ }
746
+ }
747
+
728
748
  get appendOnlyContext(): AppendOnlyContextManager | undefined {
729
749
  return this.#appendOnlyContext;
730
750
  }
@@ -902,8 +922,9 @@ export class Agent {
902
922
  this.#state.systemPrompt = typeof v === "string" ? [v] : v;
903
923
  }
904
924
 
905
- setModel(m: Model) {
906
- this.#state.model = m;
925
+ setModel(model: Model) {
926
+ this.#state.model = model;
927
+ this.#syncTokenizer(model);
907
928
  }
908
929
 
909
930
  setThinkingLevel(l: Effort | undefined) {