@oh-my-pi/pi-agent-core 17.3.7 → 17.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,111 @@
1
+ /**
2
+ * Provider-anchored transcript token accounting.
3
+ *
4
+ * Local tokenization is the expensive way to answer "how big is this
5
+ * conversation?" — and usually the wrong one, because the provider already
6
+ * answered it. Every settled assistant turn carries `usage` covering the exact
7
+ * prompt it was sent: the system prompt, the tool schemas, and every message up
8
+ * to and including itself. The only genuinely unaccounted-for text is the tail
9
+ * appended *after* that turn.
10
+ *
11
+ * These helpers locate the newest trustworthy usage report and tokenize only
12
+ * that tail, so a long session pays counting proportional to one turn instead
13
+ * of to the whole transcript, every turn.
14
+ *
15
+ * Trust rules for an anchor (mirroring the provider contract):
16
+ * - Assistant role only — nothing else carries `usage`.
17
+ * - Not `aborted` / `error`: those turns report partial or zero usage.
18
+ * - `hasContextTokenUsage(usage)`: the report must carry usable context numbers.
19
+ */
20
+
21
+ import type { AssistantMessage } from "@oh-my-pi/pi-ai";
22
+ import type { MessageCountOptions, Tokenizer } from "../tokenizer";
23
+ import type { AgentMessage } from "../types";
24
+ import { calculateContextTokens, hasContextTokenUsage } from "./compaction";
25
+
26
+ /** A provider usage report that accounts for a prefix of the transcript. */
27
+ export interface TranscriptUsageAnchor {
28
+ /** Index in the scanned array; messages at or before it are provider-accounted. */
29
+ index: number;
30
+ /** The anchoring assistant turn. */
31
+ message: AssistantMessage;
32
+ /** Conversation tokens the provider reported for that prompt. */
33
+ tokens: number;
34
+ }
35
+
36
+ /**
37
+ * Whether this message's provider usage may anchor transcript accounting.
38
+ *
39
+ * The single home for the trust rules — every anchor scan MUST route through
40
+ * it so a stale-usage rule can never drift between the transcript walkers and
41
+ * the session-entry walkers.
42
+ */
43
+ export function isTranscriptUsageAnchor(message: AgentMessage): message is AssistantMessage {
44
+ if (message.role !== "assistant") return false;
45
+ const assistant = message as AssistantMessage;
46
+ if (assistant.stopReason === "aborted" || assistant.stopReason === "error") return false;
47
+ return assistant.usage !== undefined && hasContextTokenUsage(assistant.usage);
48
+ }
49
+
50
+ /**
51
+ * Newest assistant turn in `messages[fromIndex..]` whose usage can anchor the
52
+ * transcript, or `undefined` when none qualifies (fresh context, or every
53
+ * recent turn aborted/errored).
54
+ *
55
+ * `fromIndex` excludes turns whose usage is stale — anything a compaction
56
+ * summarized away describes a prompt that is no longer sent.
57
+ */
58
+ export function findTranscriptUsageAnchor(
59
+ messages: readonly AgentMessage[],
60
+ fromIndex = 0,
61
+ ): TranscriptUsageAnchor | undefined {
62
+ for (let index = messages.length - 1; index >= fromIndex; index--) {
63
+ const message = messages[index];
64
+ if (!isTranscriptUsageAnchor(message)) continue;
65
+ return { index, message, tokens: calculateContextTokens(message.usage) };
66
+ }
67
+ return undefined;
68
+ }
69
+
70
+ /** Options for {@link estimateTranscriptTokens}. */
71
+ export interface TranscriptTokenOptions {
72
+ /**
73
+ * Gates the anchor search only: usage at or before this index is stale (a
74
+ * compaction rewrote the prompt it describes) and must not anchor. Content
75
+ * accounting is governed separately by {@link countFromIndex}.
76
+ */
77
+ anchorFromIndex?: number;
78
+ /**
79
+ * First message whose content is counted locally when no anchor is found.
80
+ * Defaults to 0 (count the whole transcript), which is what a floor
81
+ * estimate wants; pass the compaction boundary to skip summarized-away
82
+ * messages entirely.
83
+ */
84
+ countFromIndex?: number;
85
+ /** Forwarded to {@link Tokenizer.countMessage} for every locally counted message. */
86
+ excludeEncryptedReasoning?: boolean;
87
+ }
88
+
89
+ /**
90
+ * Conversation tokens for `messages`: the provider's own report for everything
91
+ * it already covers, plus a local count of only the unaccounted-for tail.
92
+ *
93
+ * An anchored result already includes the non-message prefix (system prompt +
94
+ * tool schemas) because the provider charged it; an unanchored result is a
95
+ * message-only sum. Callers that add non-message tokens on top MUST branch on
96
+ * {@link findTranscriptUsageAnchor} rather than assuming one shape.
97
+ */
98
+ export function estimateTranscriptTokens(
99
+ messages: readonly AgentMessage[],
100
+ tokenizer: Tokenizer,
101
+ options?: TranscriptTokenOptions,
102
+ ): number {
103
+ const estimateOptions: MessageCountOptions | undefined =
104
+ options?.excludeEncryptedReasoning === true ? { excludeEncryptedReasoning: true } : undefined;
105
+ const anchor = findTranscriptUsageAnchor(messages, options?.anchorFromIndex ?? 0);
106
+ let total = anchor?.tokens ?? 0;
107
+ for (let index = anchor ? anchor.index + 1 : (options?.countFromIndex ?? 0); index < messages.length; index++) {
108
+ total += tokenizer.countMessage(messages[index], estimateOptions);
109
+ }
110
+ return total;
111
+ }
@@ -208,13 +208,20 @@ export function truncateToolResultForSummary(text: string): string {
208
208
  return `${text.slice(0, TOOL_RESULT_MAX_CHARS)}\n\n[... ${truncatedChars} more characters truncated]`;
209
209
  }
210
210
 
211
+ const SUMMARY_BOUNDARY_TAG_RE = /<\s*\/?\s*(?:conversation|previous-summary)\s*>/gi;
212
+
213
+ /** Keep untrusted summary input from closing or impersonating harness-owned boundaries. */
214
+ export function escapeSummaryBoundaryTags(text: string): string {
215
+ return text.replace(SUMMARY_BOUNDARY_TAG_RE, tag => `&lt;${tag.slice(1)}`);
216
+ }
217
+
211
218
  /**
212
219
  * Serialize LLM messages as plain summary input without provider control tokens.
213
220
  */
214
221
  export function serializeConversationForSummary(messages: Message[], dialect?: Dialect): string {
215
222
  const conversation = serializeConversation(messages, dialect);
216
- if (dialect !== "harmony") return conversation;
217
- return escapeHarmonyControlTokens(conversation);
223
+ const escaped = dialect === "harmony" ? escapeHarmonyControlTokens(conversation) : conversation;
224
+ return escapeSummaryBoundaryTags(escaped);
218
225
  }
219
226
 
220
227
  /**
package/src/tokenizer.ts CHANGED
@@ -1,27 +1,282 @@
1
- import { countTokens as countTokensNat } from "@oh-my-pi/pi-natives";
1
+ import type { Model } from "@oh-my-pi/pi-ai";
2
+ import type { ModelTokenizer } from "@oh-my-pi/pi-catalog/types";
3
+ import { countTokens as countTokensNat, Encoding } from "@oh-my-pi/pi-natives";
4
+ import { stringifyJson } from "@oh-my-pi/pi-utils";
5
+ import * as snapcompact from "@oh-my-pi/snapcompact";
6
+ import { isEstimateCacheable, messageEstimateVersion } from "./compaction/message-cache";
7
+ import type { AgentMessage } from "./types";
2
8
 
3
- const accurate = process.env.PI_TOKENIZER_ACCURATE === "1" && Bun.env.NODE_ENV !== "test";
9
+ const testEnv = Bun.env.NODE_ENV === "test";
10
+ const accurate = process.env.PI_TOKENIZER_ACCURATE === "1" && !testEnv;
4
11
 
5
- function estimateTokens(text: string) {
12
+ const NATIVE_ENCODING: Record<ModelTokenizer, Encoding> = {
13
+ "claude-v3": Encoding.ClaudeV3,
14
+ "claude-v47": Encoding.ClaudeV47,
15
+ "claude-v5": Encoding.ClaudeV5,
16
+ "claude-v5-sonnet": Encoding.ClaudeV5Sonnet,
17
+ qwen3: Encoding.Qwen3,
18
+ "deepseek-v3": Encoding.DeepSeekV3,
19
+ "kimi-k2": Encoding.KimiK2,
20
+ glm5: Encoding.Glm5,
21
+ };
22
+
23
+ /** Maps the catalog-resolved tokenizer family to its native implementation. */
24
+ export function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): Encoding | null {
25
+ return model?.tokenizer ? NATIVE_ENCODING[model.tokenizer] : null;
26
+ }
27
+
28
+ /**
29
+ * `strict` always pays for an exact native count (the catalog-resolved
30
+ * tokenizer when known, o200k_base otherwise). `approximate` and
31
+ * `upperbound` prefer the same exact count for known tokenizer families or
32
+ * when `PI_TOKENIZER_ACCURATE=1` is set; otherwise they use a cheap heuristic:
33
+ * `approximate` a bytes/4 guess, `upperbound` the raw byte length (never
34
+ * undercounts).
35
+ */
36
+ export type TokenCountMode = "strict" | "approximate" | "upperbound";
37
+
38
+ /** Options for {@link Tokenizer.countMessage} / {@link Tokenizer.countMessages}. */
39
+ export interface MessageCountOptions {
40
+ /**
41
+ * Drop opaque provider reasoning payloads (`thinkingSignature`,
42
+ * `redactedThinking`, native server-tool blocks) from the estimate. Those
43
+ * are billed by the provider on replay, so the default counts them — but
44
+ * their *local* byte size can diverge wildly from what the provider
45
+ * charges, so the compaction floor (which only needs the reliably-countable,
46
+ * on-wire-compressible content) excludes them to avoid false triggers on
47
+ * thinking-heavy turns.
48
+ */
49
+ excludeEncryptedReasoning?: boolean;
50
+ }
51
+
52
+ function byteEstimate(text: string): number {
6
53
  return (Buffer.byteLength(text, "utf-8") + 3) >> 2;
7
54
  }
8
55
 
9
- export function countTokens(text: string | string[]): number {
10
- if (accurate) {
11
- return countTokensNat(text);
12
- } else if (Array.isArray(text)) {
13
- return text.reduce((sum, t) => sum + estimateTokens(t), 0);
14
- } else {
15
- return estimateTokens(text);
16
- }
56
+ function byteLength(text: string): number {
57
+ return Buffer.byteLength(text, "utf-8");
17
58
  }
18
59
 
19
- export function countTokensConservatively(text: string | string[]): number {
20
- if (accurate) {
21
- return countTokensNat(text);
22
- } else if (Array.isArray(text)) {
23
- return text.reduce((sum, value) => sum + Buffer.byteLength(value, "utf-8"), 0);
24
- } else {
25
- return Buffer.byteLength(text, "utf-8");
60
+ function sumFragments(text: string | string[], perFragment: (t: string) => number): number {
61
+ return Array.isArray(text) ? text.reduce((sum, t) => sum + perFragment(t), 0) : perFragment(text);
62
+ }
63
+
64
+ /** Verdict from {@link Tokenizer.checkTokenBudget}. */
65
+ export interface TokenBudgetCheck {
66
+ /** Whether the text fits the budget. */
67
+ fits: boolean;
68
+ /**
69
+ * Token count behind the verdict: the exact native count when `exact` is
70
+ * set, otherwise the cheap byte upper bound (which already fit, so it is
71
+ * only an over-estimate of a count known to be under budget).
72
+ */
73
+ tokens: number;
74
+ /** Whether the exact tokenizer had to run because the cheap bound busted. */
75
+ exact: boolean;
76
+ }
77
+
78
+ /**
79
+ * Image content has no tokenizer representation; charge a fixed estimate
80
+ * matching what providers typically bill for inline images.
81
+ */
82
+ const IMAGE_TOKEN_ESTIMATE = 1200;
83
+
84
+ /**
85
+ * Memoized estimates for one message under this tokenizer's encoding, split by
86
+ * the {@link MessageCountOptions.excludeEncryptedReasoning} option so the two
87
+ * variants never collide. `version` snapshots {@link messageEstimateVersion} at
88
+ * write time; an owner mutation (prune/shake/strip-images) bumps the version,
89
+ * which invalidates the entry in every live Tokenizer at once.
90
+ */
91
+ interface MessageEstimate {
92
+ version: number;
93
+ default?: number;
94
+ floored?: number;
95
+ }
96
+
97
+ /**
98
+ * Model-aware local token counter. Immutable: the catalog-resolved encoding
99
+ * is fixed at construction, so a cached count can never straddle two
100
+ * encodings. An `Agent` owns one for its active model (swapping the instance
101
+ * when the model's encoding changes); one-shot flows construct their own for
102
+ * the model that will be billed. Known tokenizer families use exact native
103
+ * counts; unknown models keep the fast byte estimate (or o200k when
104
+ * `PI_TOKENIZER_ACCURATE=1`).
105
+ */
106
+ export class Tokenizer {
107
+ readonly #encoding: Encoding | null;
108
+
109
+ /**
110
+ * Per-message estimate memo. Keyed by message identity, deliberately not a
111
+ * symbol-tagged property: callers spread messages to derive throwaway
112
+ * variants for counting (`estimateBranchSummaryTokens` does
113
+ * `countMessage({ ...message, content: truncated })`), and a property-borne
114
+ * cache would ride along the spread. Identity keying gives clones a fresh
115
+ * count.
116
+ */
117
+ #estimates = new WeakMap<AgentMessage, MessageEstimate>();
118
+
119
+ constructor(model?: Pick<Model, "tokenizer"> | null) {
120
+ this.#encoding = tokenizerEncodingForModel(model);
121
+ }
122
+
123
+ get encoding(): Encoding | null {
124
+ return this.#encoding;
125
+ }
126
+
127
+ countTokens(text: string | string[], mode: TokenCountMode = "approximate"): number {
128
+ if (mode === "strict") return countTokensNat(text, this.#encoding);
129
+ if (!testEnv && this.#encoding !== null) return countTokensNat(text, this.#encoding);
130
+ if (accurate) return countTokensNat(text);
131
+ return sumFragments(text, mode === "upperbound" ? byteLength : byteEstimate);
132
+ }
133
+
134
+ /**
135
+ * Cheap-first budget probe — the way to ask "does this fit in `budget`
136
+ * tokens?" without tokenizing the world.
137
+ *
138
+ * Byte length is a hard upper bound on token count (every token consumes at
139
+ * least one input byte), so text whose raw bytes already fit the budget
140
+ * cannot possibly exceed it — that verdict is returned without tokenizing at
141
+ * all. Only text that busts the bound is ambiguous, and only that case pays
142
+ * for the exact count. Since the bound overshoots ~4x on ordinary prose, the
143
+ * common "comfortably under budget" answer is free.
144
+ */
145
+ checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck {
146
+ const bound = sumFragments(text, byteLength);
147
+ if (bound <= budget) return { fits: true, tokens: bound, exact: false };
148
+ const tokens = this.countTokens(text, "strict");
149
+ return { fits: tokens <= budget, tokens, exact: true };
150
+ }
151
+
152
+ /**
153
+ * Token estimate for one message under this tokenizer's encoding.
154
+ *
155
+ * Settled historical messages are counted once and reused until an owner
156
+ * (prune/shake/strip-images) calls `invalidateMessageCache`; streaming
157
+ * assistants bypass the memo entirely (see the message-cache settle-gate
158
+ * invariant). Image blocks charge a fixed per-image estimate.
159
+ */
160
+ countMessage(message: AgentMessage, options?: MessageCountOptions): number {
161
+ const floored = options?.excludeEncryptedReasoning === true;
162
+ if (!isEstimateCacheable(message)) return this.#measureMessage(message, floored);
163
+ const version = messageEstimateVersion(message);
164
+ let entry = this.#estimates.get(message);
165
+ if (entry === undefined || entry.version !== version) {
166
+ entry = { version };
167
+ this.#estimates.set(message, entry);
168
+ }
169
+ const cached = floored ? entry.floored : entry.default;
170
+ if (cached !== undefined) return cached;
171
+ const result = this.#measureMessage(message, floored);
172
+ if (floored) entry.floored = result;
173
+ else entry.default = result;
174
+ return result;
175
+ }
176
+
177
+ /** Sum of {@link countMessage} over `messages`. */
178
+ countMessages(messages: readonly AgentMessage[], options?: MessageCountOptions): number {
179
+ let total = 0;
180
+ for (const message of messages) total += this.countMessage(message, options);
181
+ return total;
182
+ }
183
+
184
+ #measureMessage(message: AgentMessage, excludeEncryptedReasoning: boolean): number {
185
+ const fragments: string[] = [];
186
+ let extra = 0;
187
+ // Declaration-merged app roles (the coding-agent's bashExecution) are
188
+ // invisible to this package's union, so the discriminant is read as data.
189
+ const role: string = message.role;
190
+ if (role === "bashExecution") {
191
+ if ("command" in message && typeof message.command === "string") fragments.push(message.command);
192
+ if ("output" in message && typeof message.output === "string") fragments.push(message.output);
193
+ return fragments.length === 0 ? 0 : this.countTokens(fragments);
194
+ }
195
+
196
+ switch (message.role) {
197
+ case "user": {
198
+ const content: string | Array<{ type: string; text?: string }> = message.content;
199
+ if (typeof content === "string") {
200
+ fragments.push(content);
201
+ } else if (Array.isArray(content)) {
202
+ for (const block of content) {
203
+ if (block.type === "text" && block.text) {
204
+ fragments.push(block.text);
205
+ }
206
+ }
207
+ }
208
+ break;
209
+ }
210
+ case "assistant": {
211
+ for (const block of message.content) {
212
+ if (block.type === "text") {
213
+ fragments.push(block.text);
214
+ } else if (block.type === "thinking") {
215
+ fragments.push(block.thinking);
216
+ // Providers charge for the opaque signature/reasoning payload that
217
+ // rides alongside the thinking text (OpenAI Responses encrypted
218
+ // reasoning items, Anthropic signed thinking blocks, etc.). Without
219
+ // counting it, this estimator can read ~half of the provider-reported
220
+ // usage on thinking-heavy turns — see #2275 for the resulting
221
+ // compaction-trigger / post-check metric divergence. The compaction
222
+ // floor excludes it (its local byte size diverges from provider billing).
223
+ if (block.thinkingSignature && !excludeEncryptedReasoning) {
224
+ fragments.push(block.thinkingSignature);
225
+ }
226
+ } else if (block.type === "toolCall") {
227
+ fragments.push(block.name);
228
+ fragments.push(stringifyJson(block.arguments) ?? "null");
229
+ } else if (block.type === "redactedThinking") {
230
+ // Encrypted reasoning blob the provider still bills for on replay;
231
+ // excluded from the compaction floor for the same reason as above.
232
+ if (!excludeEncryptedReasoning) fragments.push(block.data);
233
+ } else if (block.type === "anthropicServerTool") {
234
+ // Native Anthropic server-tool call/result replayed verbatim on the
235
+ // wire (server_tool_use input and opaque result content). The provider
236
+ // still bills for it on same-provider replay; excluded from the
237
+ // compaction floor like other encrypted reasoning because its local
238
+ // byte size diverges from provider billing.
239
+ if (!excludeEncryptedReasoning) fragments.push(stringifyJson(block.block) ?? "null");
240
+ }
241
+ }
242
+ break;
243
+ }
244
+ case "hookMessage":
245
+ case "toolResult": {
246
+ if (typeof message.content === "string") {
247
+ fragments.push(message.content);
248
+ } else {
249
+ for (const block of message.content) {
250
+ if (block.type === "text" && block.text) {
251
+ fragments.push(block.text);
252
+ } else if (block.type === "image") {
253
+ extra += IMAGE_TOKEN_ESTIMATE;
254
+ }
255
+ }
256
+ }
257
+ break;
258
+ }
259
+ case "branchSummary":
260
+ case "compactionSummary": {
261
+ fragments.push(message.summary);
262
+ if (message.role === "compactionSummary") {
263
+ if (message.blocks) {
264
+ for (const block of message.blocks) {
265
+ if (block.type === "text") fragments.push(block.text);
266
+ else extra += snapcompact.FRAME_TOKEN_ESTIMATE;
267
+ }
268
+ } else if (message.images) {
269
+ // Snapcompact frames render at ≥1568px; providers bill the downscaled cap.
270
+ extra += message.images.length * snapcompact.FRAME_TOKEN_ESTIMATE;
271
+ }
272
+ }
273
+ break;
274
+ }
275
+ default:
276
+ return 0;
277
+ }
278
+
279
+ if (fragments.length === 0) return extra;
280
+ return extra + this.countTokens(fragments);
26
281
  }
27
282
  }
package/src/types.ts CHANGED
@@ -208,7 +208,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
208
208
  * @example
209
209
  * ```typescript
210
210
  * transformContext: async (messages) => {
211
- * if (estimateTokens(messages) > MAX_TOKENS) {
211
+ * if (agent.tokenizer.countMessages(messages) > MAX_TOKENS) {
212
212
  * return pruneOldMessages(messages);
213
213
  * }
214
214
  * return messages;