@oh-my-pi/pi-agent-core 17.3.7 → 17.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/dist/types/agent.d.ts +8 -1
- package/dist/types/compaction/branch-summarization.d.ts +2 -1
- package/dist/types/compaction/compaction.d.ts +37 -17
- package/dist/types/compaction/entries.d.ts +10 -1
- package/dist/types/compaction/index.d.ts +1 -0
- package/dist/types/compaction/message-cache.d.ts +6 -4
- package/dist/types/compaction/messages.d.ts +17 -1
- package/dist/types/compaction/openai.d.ts +2 -1
- package/dist/types/compaction/pruning.d.ts +3 -2
- package/dist/types/compaction/shake.d.ts +2 -1
- package/dist/types/compaction/transcript-tokens.d.ts +76 -0
- package/dist/types/compaction/utils.d.ts +2 -0
- package/dist/types/tokenizer.d.ts +78 -2
- package/dist/types/types.d.ts +1 -1
- package/package.json +8 -8
- package/src/agent.ts +25 -4
- package/src/compaction/branch-summarization.ts +15 -8
- package/src/compaction/compaction.ts +234 -162
- package/src/compaction/entries.ts +11 -0
- package/src/compaction/index.ts +1 -0
- package/src/compaction/message-cache.ts +33 -34
- package/src/compaction/messages.ts +21 -5
- package/src/compaction/openai.ts +52 -20
- package/src/compaction/prompts/summarization-system.md +2 -0
- package/src/compaction/pruning.ts +25 -13
- package/src/compaction/shake.ts +18 -16
- package/src/compaction/transcript-tokens.ts +111 -0
- package/src/compaction/utils.ts +9 -2
- package/src/tokenizer.ts +273 -18
- package/src/types.ts +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,33 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [17.4.0] - 2026-08-20
|
|
6
|
+
|
|
7
|
+
### Breaking Changes
|
|
8
|
+
|
|
9
|
+
- Replaced global token counting functions (`countTokens`, `countTokensConservatively`, `setTokenizerModel`, and `estimateTokens`) with model-scoped, immutable `Tokenizer` instances (`agent.tokenizer`). Use `tokenizer.countTokens(text, mode?)`, `tokenizer.countMessage(message)`, or `tokenizer.countMessages(messages)`.
|
|
10
|
+
- Updated context management functions (`findCutPoint`, `prepareBranchEntries`, `collectShakeRegions`, `pruneToolOutputs`, `pruneSupersededToolResults`, and `trimRemoteCompactionInputToContextWindow`) to require an explicit `Tokenizer` instance.
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
|
|
14
|
+
- Added `Tokenizer.checkTokenBudget(text, budget)` to efficiently verify if text fits within a token limit using fast byte-bound checks before falling back to full tokenization.
|
|
15
|
+
- Added provider-anchored transcript token estimation (`findTranscriptUsageAnchor`, `isTranscriptUsageAnchor`, `estimateTranscriptTokens`) to calculate transcript token counts incrementally from the latest reported assistant turn usage.
|
|
16
|
+
- Added `remotePreserveReusable()` to check whether a previous remote compaction payload remains reusable with the active model.
|
|
17
|
+
|
|
18
|
+
### Changed
|
|
19
|
+
|
|
20
|
+
- Expanded native tokenizer support across catalog models, adding exact embedded token counting for Claude, Qwen 3.5+, DeepSeek V3/V4/R1, Kimi K2/K3, and GLM-5+ models. `Tokenizer` now constructs from a resolved catalog `Model`.
|
|
21
|
+
- `createCompactionSummaryMessage` takes an options object after `(summary, tokensBefore, timestamp)`; `CompactionSummaryMessage` gained optional `method` and `tokensAfter` display metadata.
|
|
22
|
+
|
|
23
|
+
## [17.3.8] - 2026-08-19
|
|
24
|
+
|
|
25
|
+
### Fixed
|
|
26
|
+
|
|
27
|
+
- Fixed `/compact` (and automatic compaction) resurrecting pre-`/clear` conversation turns: `prepareCompaction` now honors the latest `reset_boundary`, so a compaction after an in-place `/clear` only summarizes messages created after the reset ([#8718](https://github.com/can1357/oh-my-pi/issues/8718)).
|
|
28
|
+
- Hardened compaction summarization against prompt injection: conversation history and previous summaries are now treated as untrusted, and embedded `<conversation>`/`<previous-summary>` boundary tags are neutralized before prompt assembly ([#8727](https://github.com/can1357/oh-my-pi/pull/8727) by [@koopmannleon19977-cmyk](https://github.com/koopmannleon19977-cmyk)).
|
|
29
|
+
- Compaction summarization input is now bounded to the summary model's context (windowed fold for oversized spans) and deterministic context-overflow 400s are no longer retried up to the full retry budget; artifact ids containing `503` no longer misclassify hard 400s as transient.
|
|
30
|
+
- Fixed remote compaction mirroring the #8789 Responses shape: `buildOpenAiNativeHistory` now hoists an assistant `message` wedged between a tool-call batch and its outputs ahead of the batch, so compaction requests to strict opencode-go gateways match the canonical `message(s) → calls → outputs` order ([#8789](https://github.com/can1357/oh-my-pi/issues/8789)).
|
|
31
|
+
|
|
5
32
|
## [17.3.5] - 2026-08-16
|
|
6
33
|
|
|
7
34
|
### Added
|
package/dist/types/agent.d.ts
CHANGED
|
@@ -2,6 +2,7 @@ import { type ApiKey, type AssistantMessage, type AssistantMessageEvent, type Co
|
|
|
2
2
|
import type { Dialect } from "@oh-my-pi/pi-ai/dialect";
|
|
3
3
|
import type { HarmonyAuditEvent } from "@oh-my-pi/pi-ai/utils/harmony-leak";
|
|
4
4
|
import type { AppendOnlyContextManager } from "./append-only-context.js";
|
|
5
|
+
import { Tokenizer } from "./tokenizer.js";
|
|
5
6
|
import type { AgentBeforeModelCall, AgentEvent, AgentLoopConfig, AgentMessage, AgentState, AgentTool, AgentToolContext, AgentTurnEndContext, AsideMessage, StreamFn, ToolCallContext, ToolChoiceDirective } from "./types.js";
|
|
6
7
|
export declare class AgentBusyError extends Error {
|
|
7
8
|
constructor(message?: string);
|
|
@@ -354,6 +355,12 @@ export declare class Agent {
|
|
|
354
355
|
*/
|
|
355
356
|
set maxRetryDelayMs(value: number | undefined);
|
|
356
357
|
get state(): AgentState;
|
|
358
|
+
/**
|
|
359
|
+
* Tokenizer for the active model. The instance is replaced whenever the
|
|
360
|
+
* active model's encoding changes (see {@link setModel}), so callers must
|
|
361
|
+
* not cache it across model switches.
|
|
362
|
+
*/
|
|
363
|
+
get tokenizer(): Tokenizer;
|
|
357
364
|
get appendOnlyContext(): AppendOnlyContextManager | undefined;
|
|
358
365
|
setAppendOnlyContext(manager?: AppendOnlyContextManager): void;
|
|
359
366
|
/**
|
|
@@ -401,7 +408,7 @@ export declare class Agent {
|
|
|
401
408
|
setAsideMessageProvider(fn: (() => AsideMessage[] | Promise<AsideMessage[]>) | undefined): void;
|
|
402
409
|
emitExternalEvent(event: AgentEvent): void;
|
|
403
410
|
setSystemPrompt(v: string[] | string): void;
|
|
404
|
-
setModel(
|
|
411
|
+
setModel(model: Model): void;
|
|
405
412
|
setThinkingLevel(l: Effort | undefined): void;
|
|
406
413
|
setDisableReasoning(disabled: boolean): void;
|
|
407
414
|
setSteeringMode(mode: "all" | "one-at-a-time"): void;
|
|
@@ -6,6 +6,7 @@
|
|
|
6
6
|
*/
|
|
7
7
|
import type { Api, ApiKey, AssistantMessage, Context, Model, SimpleStreamOptions } from "@oh-my-pi/pi-ai";
|
|
8
8
|
import { type AgentTelemetry } from "../telemetry.js";
|
|
9
|
+
import { Tokenizer } from "../tokenizer.js";
|
|
9
10
|
import type { AgentMessage } from "../types.js";
|
|
10
11
|
import type { ReadonlySessionManager, SessionEntry } from "./entries.js";
|
|
11
12
|
import { type ConvertToLlm } from "./messages.js";
|
|
@@ -91,7 +92,7 @@ export declare function collectEntriesForBranchSummary(session: ReadonlySessionM
|
|
|
91
92
|
* @param entries - Entries in chronological order
|
|
92
93
|
* @param tokenBudget - Maximum tokens to include (0 = no limit)
|
|
93
94
|
*/
|
|
94
|
-
export declare function prepareBranchEntries(entries: SessionEntry[], tokenBudget?: number): BranchPreparation;
|
|
95
|
+
export declare function prepareBranchEntries(entries: SessionEntry[], tokenizer: Tokenizer, tokenBudget?: number): BranchPreparation;
|
|
95
96
|
/**
|
|
96
97
|
* Generate a summary of abandoned branch entries.
|
|
97
98
|
*
|
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
import { type Api, type ApiKey, type AssistantMessage, type CodexCompactionContext, type Context, type FetchImpl, type MessageAttribution, type Model, type OneshotRetryOptions, type ProviderSessionState, type SimpleStreamOptions, type Tool, type Usage } from "@oh-my-pi/pi-ai";
|
|
8
8
|
import { type AgentTelemetry } from "../telemetry.js";
|
|
9
9
|
import { ThinkingLevel } from "../thinking.js";
|
|
10
|
+
import { Tokenizer } from "../tokenizer.js";
|
|
10
11
|
import type { AgentMessage } from "../types.js";
|
|
11
12
|
import type { SessionEntry } from "./entries.js";
|
|
12
13
|
import { type ConvertToLlm } from "./messages.js";
|
|
@@ -119,21 +120,6 @@ export declare function shouldCompact(contextTokens: number, contextWindow: numb
|
|
|
119
120
|
*/
|
|
120
121
|
export declare function compactionContextTokens(providerContextTokens: number, storedConversationEstimate: number): number;
|
|
121
122
|
export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings): number;
|
|
122
|
-
/**
|
|
123
|
-
* Estimate token count for a message using cl100k_base via the native
|
|
124
|
-
* tokenizer. This is not Claude's first-party tokenizer (Anthropic doesn't
|
|
125
|
-
* publish one) but is within ~5–10% across English/code text.
|
|
126
|
-
*
|
|
127
|
-
* `excludeEncryptedReasoning` drops opaque provider reasoning payloads
|
|
128
|
-
* (`thinkingSignature`, `redactedThinking`) from the estimate. Those are billed
|
|
129
|
-
* by the provider on replay, so the default counts them — but their *local*
|
|
130
|
-
* byte size can diverge wildly from what the provider charges, so the
|
|
131
|
-
* compaction floor (which only needs the reliably-countable, on-wire-compressible
|
|
132
|
-
* content) excludes them to avoid false triggers on thinking-heavy turns.
|
|
133
|
-
*/
|
|
134
|
-
export declare function estimateTokens(message: AgentMessage, options?: {
|
|
135
|
-
excludeEncryptedReasoning?: boolean;
|
|
136
|
-
}): number;
|
|
137
123
|
/**
|
|
138
124
|
* Find the user message (or bashExecution) that starts the turn containing the given entry index.
|
|
139
125
|
* Returns -1 if no turn start found before the index.
|
|
@@ -164,7 +150,7 @@ export interface CutPointResult {
|
|
|
164
150
|
*
|
|
165
151
|
* Only considers entries between `startIndex` and `endIndex` (exclusive).
|
|
166
152
|
*/
|
|
167
|
-
export declare function findCutPoint(entries: SessionEntry[], startIndex: number, endIndex: number, keepRecentTokens: number): CutPointResult;
|
|
153
|
+
export declare function findCutPoint(entries: SessionEntry[], tokenizer: Tokenizer, startIndex: number, endIndex: number, keepRecentTokens: number): CutPointResult;
|
|
168
154
|
export declare const AUTO_HANDOFF_THRESHOLD_FOCUS: string;
|
|
169
155
|
/**
|
|
170
156
|
* Generate a summary of the conversation using the LLM.
|
|
@@ -310,7 +296,41 @@ export interface CompactionPreparation {
|
|
|
310
296
|
/** Compaction settions from settings.jsonl */
|
|
311
297
|
settings: CompactionSettings;
|
|
312
298
|
}
|
|
313
|
-
|
|
299
|
+
/**
|
|
300
|
+
* Whether a prior remote compaction's provider-native replay can still be read
|
|
301
|
+
* by the active model — the model that assembles the request context on every
|
|
302
|
+
* turn. A local compaction (no remote preserve) always can: it holds a real
|
|
303
|
+
* textual summary. A remote compaction (V2 or V1) only can when the active model
|
|
304
|
+
* shares the blob's provider AND remote replay is still enabled; otherwise the
|
|
305
|
+
* active model's encoder drops the payload (see `getOpenAIResponsesHistoryPayload`)
|
|
306
|
+
* and only the opaque placeholder summary survives, so the caller must re-expand
|
|
307
|
+
* the originals into a portable local summary rather than strand that history.
|
|
308
|
+
*
|
|
309
|
+
* Judged against the ACTIVE model, not the compaction candidate set: a role
|
|
310
|
+
* model (e.g. `modelRoles.smol`) that still maps to the blob's provider does not
|
|
311
|
+
* let the active model replay it, so keying reuse on "any candidate shares the
|
|
312
|
+
* provider" left a provider-switched session permanently context-less (#6343).
|
|
313
|
+
*/
|
|
314
|
+
export declare function remotePreserveReusable(preserveData: Record<string, unknown> | undefined, activeModel: Model, settings: CompactionSettings): boolean;
|
|
315
|
+
/**
|
|
316
|
+
* Index of the newest compaction entry the active model can actually read, or
|
|
317
|
+
* `-1` when none can.
|
|
318
|
+
*
|
|
319
|
+
* A provider-native remote compaction (V2 or V1) stores an opaque replay payload
|
|
320
|
+
* and only a placeholder summary, so for any OTHER provider that entry
|
|
321
|
+
* summarizes nothing and the history behind it is still live context. Callers
|
|
322
|
+
* must therefore treat it as absent: `prepareCompaction` re-expands past it and
|
|
323
|
+
* summarizes those messages locally, and the maintenance ops that use the
|
|
324
|
+
* compaction boundary to skip "already summarized away" entries must not skip
|
|
325
|
+
* entries that no summary covers.
|
|
326
|
+
*/
|
|
327
|
+
export declare function findReadableCompactionIndex(pathEntries: SessionEntry[], settings: CompactionSettings, activeModel?: Model): number;
|
|
328
|
+
/**
|
|
329
|
+
* Pass the caller's warm `tokenizer` (the Agent's for the active model) so the
|
|
330
|
+
* full-branch estimate walk hits its memo; the cold default is for one-shot
|
|
331
|
+
* callers that have no live agent.
|
|
332
|
+
*/
|
|
333
|
+
export declare function prepareCompaction(pathEntries: SessionEntry[], settings: CompactionSettings, activeModel?: Model, tokenizer?: Tokenizer): CompactionPreparation | undefined;
|
|
314
334
|
/**
|
|
315
335
|
* Generate summaries for compaction using prepared data.
|
|
316
336
|
* Returns CompactionResult - SessionManager adds id/parentId when saving.
|
|
@@ -102,9 +102,18 @@ export interface ModeChangeEntry extends SessionEntryBase {
|
|
|
102
102
|
/** Optional mode-specific data (e.g. plan file path) */
|
|
103
103
|
data?: Record<string, unknown>;
|
|
104
104
|
}
|
|
105
|
+
/**
|
|
106
|
+
* Durable context-reset marker recorded by an in-place `/clear`. It carries no
|
|
107
|
+
* payload — its presence on the branch means every entry before it was dropped
|
|
108
|
+
* from the model context, so context assembly and compaction start after the
|
|
109
|
+
* latest one. The full pre-reset history stays on disk for transcript export.
|
|
110
|
+
*/
|
|
111
|
+
export interface ResetBoundaryEntry extends SessionEntryBase {
|
|
112
|
+
type: "reset_boundary";
|
|
113
|
+
}
|
|
105
114
|
export interface CustomCompactionSessionEntries {
|
|
106
115
|
}
|
|
107
|
-
export type SessionEntry = SessionMessageEntry | ThinkingLevelChangeEntry | ModelChangeEntry | ServiceTierChangeEntry | CompactionEntry | BranchSummaryEntry | CustomEntry | CustomMessageEntry | LabelEntry | TitleChangeEntry | TtsrInjectionEntry | SessionInitEntry | ModeChangeEntry | CustomCompactionSessionEntries[keyof CustomCompactionSessionEntries];
|
|
116
|
+
export type SessionEntry = SessionMessageEntry | ThinkingLevelChangeEntry | ModelChangeEntry | ServiceTierChangeEntry | CompactionEntry | BranchSummaryEntry | CustomEntry | CustomMessageEntry | LabelEntry | TitleChangeEntry | TtsrInjectionEntry | SessionInitEntry | ModeChangeEntry | ResetBoundaryEntry | CustomCompactionSessionEntries[keyof CustomCompactionSessionEntries];
|
|
108
117
|
export interface ReadonlySessionManager {
|
|
109
118
|
getBranch(leafId?: string | null): SessionEntry[];
|
|
110
119
|
getEntry(id: string): SessionEntry | undefined;
|
|
@@ -5,16 +5,18 @@ import type { AgentMessage } from "../types.js";
|
|
|
5
5
|
* unregister function. The coding-agent `convertToLlm` memo registers here.
|
|
6
6
|
*/
|
|
7
7
|
export declare function registerMessageCacheInvalidator(invalidate: (message: AgentMessage) => void): () => void;
|
|
8
|
+
/**
|
|
9
|
+
* Current estimate version of `message` (0 until first invalidation). A
|
|
10
|
+
* `Tokenizer` memo entry stamped with an older version is stale and must be
|
|
11
|
+
* recounted.
|
|
12
|
+
*/
|
|
13
|
+
export declare function messageEstimateVersion(message: AgentMessage): number;
|
|
8
14
|
/**
|
|
9
15
|
* True when this message's estimate is safe to cache by identity. Non-assistants
|
|
10
16
|
* are immutable once appended; assistants are cached only once settled (see the
|
|
11
17
|
* settle-gate invariant above).
|
|
12
18
|
*/
|
|
13
19
|
export declare function isEstimateCacheable(message: AgentMessage): boolean;
|
|
14
|
-
/** Read a cached estimate for the given option split, or `undefined` on miss. */
|
|
15
|
-
export declare function readEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean): number | undefined;
|
|
16
|
-
/** Store an estimate for the given option split. */
|
|
17
|
-
export declare function writeEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean, value: number): void;
|
|
18
20
|
/**
|
|
19
21
|
* Drop every cached derivation of `message` after an in-place rewrite. Owners of
|
|
20
22
|
* mutation (prune, shake, strip-images) call this at the mutation seam so the
|
|
@@ -32,6 +32,10 @@ export interface CompactionSummaryMessage {
|
|
|
32
32
|
summary: string;
|
|
33
33
|
shortSummary?: string;
|
|
34
34
|
tokensBefore: number;
|
|
35
|
+
/** Estimated context tokens after the rewrite (display metadata). */
|
|
36
|
+
tokensAfter?: number;
|
|
37
|
+
/** Harness compaction method that produced this summary (display metadata). */
|
|
38
|
+
method?: string;
|
|
35
39
|
providerPayload?: ProviderPayload;
|
|
36
40
|
/** Runtime-only ordered archive blocks for snapcompact: old text region,
|
|
37
41
|
* imaged middle, then new text region. When present, `summary` is already
|
|
@@ -56,7 +60,19 @@ export type ConvertToLlm = (messages: AgentMessage[]) => Message[];
|
|
|
56
60
|
export declare function renderBranchSummaryContext(summary: string): string;
|
|
57
61
|
export declare function renderCompactionSummaryContext(summary: string): string;
|
|
58
62
|
export declare function createBranchSummaryMessage(summary: string, fromId: string, timestamp: string): BranchSummaryMessage;
|
|
59
|
-
|
|
63
|
+
/** Optional metadata for {@link createCompactionSummaryMessage}. */
|
|
64
|
+
export interface CompactionSummaryMessageOptions {
|
|
65
|
+
shortSummary?: string;
|
|
66
|
+
providerPayload?: ProviderPayload;
|
|
67
|
+
images?: ImageContent[];
|
|
68
|
+
blocks?: (TextContent | ImageContent)[];
|
|
69
|
+
warning?: string;
|
|
70
|
+
/** Harness compaction method that produced this summary (e.g. "remote", "soft", "handoff"). */
|
|
71
|
+
method?: string;
|
|
72
|
+
/** Estimated context tokens after the rewrite, for display alongside `tokensBefore`. */
|
|
73
|
+
tokensAfter?: number;
|
|
74
|
+
}
|
|
75
|
+
export declare function createCompactionSummaryMessage(summary: string, tokensBefore: number, timestamp: string, options?: CompactionSummaryMessageOptions): CompactionSummaryMessage;
|
|
60
76
|
export declare function createCustomMessage(customType: string, content: string | (TextContent | ImageContent)[], display: boolean, details: unknown | undefined, timestamp: string, attribution?: MessageAttribution): CustomMessage;
|
|
61
77
|
/**
|
|
62
78
|
* Transform a single core-domain agent message to its LLM form; `undefined`
|
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
* with `{ summary, shortSummary? }`.
|
|
16
16
|
*/
|
|
17
17
|
import type { CodexCompactionContext, FetchImpl, Message, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
|
|
18
|
+
import { Tokenizer } from "../tokenizer.js";
|
|
18
19
|
export * from "./compaction-v2-streaming.js";
|
|
19
20
|
export declare const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
|
|
20
21
|
/**
|
|
@@ -39,7 +40,7 @@ export interface TrimRemoteCompactionInputResult {
|
|
|
39
40
|
* outputs keeps call/result pairing and all earlier assistant/reasoning history,
|
|
40
41
|
* matching Codex's recovery path for oversized tool turns.
|
|
41
42
|
*/
|
|
42
|
-
export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
|
|
43
|
+
export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, tokenizer: Tokenizer, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
|
|
43
44
|
export type OpenAiRemoteCompactionItem = {
|
|
44
45
|
type: "compaction" | "compaction_summary";
|
|
45
46
|
encrypted_content?: string;
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Tool output pruning utilities for compaction.
|
|
3
3
|
*/
|
|
4
|
+
import type { Tokenizer } from "../tokenizer.js";
|
|
4
5
|
import type { SessionEntry } from "./entries.js";
|
|
5
6
|
import { type ProtectedToolMatcher } from "./tool-protection.js";
|
|
6
7
|
export interface PruneConfig {
|
|
@@ -90,8 +91,8 @@ export interface SupersedePruneConfig {
|
|
|
90
91
|
* the provider cache is cold anyway (then all still-sent candidates flush).
|
|
91
92
|
* Never mutates entries before `keepBoundaryId` (summarized away — not sent).
|
|
92
93
|
*/
|
|
93
|
-
export declare function pruneSupersededToolResults(entries: SessionEntry[], config: SupersedePruneConfig): PruneResult;
|
|
94
|
-
export declare function pruneToolOutputs(entries: SessionEntry[], config?: PruneConfig): PruneResult;
|
|
94
|
+
export declare function pruneSupersededToolResults(entries: SessionEntry[], tokenizer: Tokenizer, config: SupersedePruneConfig): PruneResult;
|
|
95
|
+
export declare function pruneToolOutputs(entries: SessionEntry[], tokenizer: Tokenizer, config?: PruneConfig): PruneResult;
|
|
95
96
|
/**
|
|
96
97
|
* Supersede key for the `read` tool: the file path with the trailing line/raw
|
|
97
98
|
* selector stripped (the read tool's own splitter grammar via
|
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
*
|
|
10
10
|
* Layering mirrors `pruning.ts`: no I/O here.
|
|
11
11
|
*/
|
|
12
|
+
import type { Tokenizer } from "../tokenizer.js";
|
|
12
13
|
import type { CustomMessageEntry, SessionEntry, SessionMessageEntry } from "./entries.js";
|
|
13
14
|
import { type ProtectedToolMatcher } from "./tool-protection.js";
|
|
14
15
|
export interface ShakeConfig {
|
|
@@ -77,7 +78,7 @@ export type ShakeRegion = ToolResultShakeRegion | BlockShakeRegion;
|
|
|
77
78
|
* and regions never span a message boundary. When the combined estimated
|
|
78
79
|
* savings is below `minSavings`, returns `[]` (no-op).
|
|
79
80
|
*/
|
|
80
|
-
export declare function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig): ShakeRegion[];
|
|
81
|
+
export declare function collectShakeRegions(entries: SessionEntry[], tokenizer: Tokenizer, config: ShakeConfig): ShakeRegion[];
|
|
81
82
|
/**
|
|
82
83
|
* Pure mutation: replace a single region's content in place.
|
|
83
84
|
*
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Provider-anchored transcript token accounting.
|
|
3
|
+
*
|
|
4
|
+
* Local tokenization is the expensive way to answer "how big is this
|
|
5
|
+
* conversation?" — and usually the wrong one, because the provider already
|
|
6
|
+
* answered it. Every settled assistant turn carries `usage` covering the exact
|
|
7
|
+
* prompt it was sent: the system prompt, the tool schemas, and every message up
|
|
8
|
+
* to and including itself. The only genuinely unaccounted-for text is the tail
|
|
9
|
+
* appended *after* that turn.
|
|
10
|
+
*
|
|
11
|
+
* These helpers locate the newest trustworthy usage report and tokenize only
|
|
12
|
+
* that tail, so a long session pays counting proportional to one turn instead
|
|
13
|
+
* of to the whole transcript, every turn.
|
|
14
|
+
*
|
|
15
|
+
* Trust rules for an anchor (mirroring the provider contract):
|
|
16
|
+
* - Assistant role only — nothing else carries `usage`.
|
|
17
|
+
* - Not `aborted` / `error`: those turns report partial or zero usage.
|
|
18
|
+
* - `hasContextTokenUsage(usage)`: the report must carry usable context numbers.
|
|
19
|
+
*/
|
|
20
|
+
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
|
|
21
|
+
import type { Tokenizer } from "../tokenizer.js";
|
|
22
|
+
import type { AgentMessage } from "../types.js";
|
|
23
|
+
/** A provider usage report that accounts for a prefix of the transcript. */
|
|
24
|
+
export interface TranscriptUsageAnchor {
|
|
25
|
+
/** Index in the scanned array; messages at or before it are provider-accounted. */
|
|
26
|
+
index: number;
|
|
27
|
+
/** The anchoring assistant turn. */
|
|
28
|
+
message: AssistantMessage;
|
|
29
|
+
/** Conversation tokens the provider reported for that prompt. */
|
|
30
|
+
tokens: number;
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* Whether this message's provider usage may anchor transcript accounting.
|
|
34
|
+
*
|
|
35
|
+
* The single home for the trust rules — every anchor scan MUST route through
|
|
36
|
+
* it so a stale-usage rule can never drift between the transcript walkers and
|
|
37
|
+
* the session-entry walkers.
|
|
38
|
+
*/
|
|
39
|
+
export declare function isTranscriptUsageAnchor(message: AgentMessage): message is AssistantMessage;
|
|
40
|
+
/**
|
|
41
|
+
* Newest assistant turn in `messages[fromIndex..]` whose usage can anchor the
|
|
42
|
+
* transcript, or `undefined` when none qualifies (fresh context, or every
|
|
43
|
+
* recent turn aborted/errored).
|
|
44
|
+
*
|
|
45
|
+
* `fromIndex` excludes turns whose usage is stale — anything a compaction
|
|
46
|
+
* summarized away describes a prompt that is no longer sent.
|
|
47
|
+
*/
|
|
48
|
+
export declare function findTranscriptUsageAnchor(messages: readonly AgentMessage[], fromIndex?: number): TranscriptUsageAnchor | undefined;
|
|
49
|
+
/** Options for {@link estimateTranscriptTokens}. */
|
|
50
|
+
export interface TranscriptTokenOptions {
|
|
51
|
+
/**
|
|
52
|
+
* Gates the anchor search only: usage at or before this index is stale (a
|
|
53
|
+
* compaction rewrote the prompt it describes) and must not anchor. Content
|
|
54
|
+
* accounting is governed separately by {@link countFromIndex}.
|
|
55
|
+
*/
|
|
56
|
+
anchorFromIndex?: number;
|
|
57
|
+
/**
|
|
58
|
+
* First message whose content is counted locally when no anchor is found.
|
|
59
|
+
* Defaults to 0 (count the whole transcript), which is what a floor
|
|
60
|
+
* estimate wants; pass the compaction boundary to skip summarized-away
|
|
61
|
+
* messages entirely.
|
|
62
|
+
*/
|
|
63
|
+
countFromIndex?: number;
|
|
64
|
+
/** Forwarded to {@link Tokenizer.countMessage} for every locally counted message. */
|
|
65
|
+
excludeEncryptedReasoning?: boolean;
|
|
66
|
+
}
|
|
67
|
+
/**
|
|
68
|
+
* Conversation tokens for `messages`: the provider's own report for everything
|
|
69
|
+
* it already covers, plus a local count of only the unaccounted-for tail.
|
|
70
|
+
*
|
|
71
|
+
* An anchored result already includes the non-message prefix (system prompt +
|
|
72
|
+
* tool schemas) because the provider charged it; an unanchored result is a
|
|
73
|
+
* message-only sum. Callers that add non-message tokens on top MUST branch on
|
|
74
|
+
* {@link findTranscriptUsageAnchor} rather than assuming one shape.
|
|
75
|
+
*/
|
|
76
|
+
export declare function estimateTranscriptTokens(messages: readonly AgentMessage[], tokenizer: Tokenizer, options?: TranscriptTokenOptions): number;
|
|
@@ -49,6 +49,8 @@ export declare function upsertFileOperations(summary: string, readFiles: string[
|
|
|
49
49
|
* Truncate tool results to the same representation used in summarization prompts.
|
|
50
50
|
*/
|
|
51
51
|
export declare function truncateToolResultForSummary(text: string): string;
|
|
52
|
+
/** Keep untrusted summary input from closing or impersonating harness-owned boundaries. */
|
|
53
|
+
export declare function escapeSummaryBoundaryTags(text: string): string;
|
|
52
54
|
/**
|
|
53
55
|
* Serialize LLM messages as plain summary input without provider control tokens.
|
|
54
56
|
*/
|
|
@@ -1,2 +1,78 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
1
|
+
import type { Model } from "@oh-my-pi/pi-ai";
|
|
2
|
+
import { Encoding } from "@oh-my-pi/pi-natives";
|
|
3
|
+
import type { AgentMessage } from "./types.js";
|
|
4
|
+
/** Maps the catalog-resolved tokenizer family to its native implementation. */
|
|
5
|
+
export declare function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): Encoding | null;
|
|
6
|
+
/**
|
|
7
|
+
* `strict` always pays for an exact native count (the catalog-resolved
|
|
8
|
+
* tokenizer when known, o200k_base otherwise). `approximate` and
|
|
9
|
+
* `upperbound` prefer the same exact count for known tokenizer families or
|
|
10
|
+
* when `PI_TOKENIZER_ACCURATE=1` is set; otherwise they use a cheap heuristic:
|
|
11
|
+
* `approximate` a bytes/4 guess, `upperbound` the raw byte length (never
|
|
12
|
+
* undercounts).
|
|
13
|
+
*/
|
|
14
|
+
export type TokenCountMode = "strict" | "approximate" | "upperbound";
|
|
15
|
+
/** Options for {@link Tokenizer.countMessage} / {@link Tokenizer.countMessages}. */
|
|
16
|
+
export interface MessageCountOptions {
|
|
17
|
+
/**
|
|
18
|
+
* Drop opaque provider reasoning payloads (`thinkingSignature`,
|
|
19
|
+
* `redactedThinking`, native server-tool blocks) from the estimate. Those
|
|
20
|
+
* are billed by the provider on replay, so the default counts them — but
|
|
21
|
+
* their *local* byte size can diverge wildly from what the provider
|
|
22
|
+
* charges, so the compaction floor (which only needs the reliably-countable,
|
|
23
|
+
* on-wire-compressible content) excludes them to avoid false triggers on
|
|
24
|
+
* thinking-heavy turns.
|
|
25
|
+
*/
|
|
26
|
+
excludeEncryptedReasoning?: boolean;
|
|
27
|
+
}
|
|
28
|
+
/** Verdict from {@link Tokenizer.checkTokenBudget}. */
|
|
29
|
+
export interface TokenBudgetCheck {
|
|
30
|
+
/** Whether the text fits the budget. */
|
|
31
|
+
fits: boolean;
|
|
32
|
+
/**
|
|
33
|
+
* Token count behind the verdict: the exact native count when `exact` is
|
|
34
|
+
* set, otherwise the cheap byte upper bound (which already fit, so it is
|
|
35
|
+
* only an over-estimate of a count known to be under budget).
|
|
36
|
+
*/
|
|
37
|
+
tokens: number;
|
|
38
|
+
/** Whether the exact tokenizer had to run because the cheap bound busted. */
|
|
39
|
+
exact: boolean;
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Model-aware local token counter. Immutable: the catalog-resolved encoding
|
|
43
|
+
* is fixed at construction, so a cached count can never straddle two
|
|
44
|
+
* encodings. An `Agent` owns one for its active model (swapping the instance
|
|
45
|
+
* when the model's encoding changes); one-shot flows construct their own for
|
|
46
|
+
* the model that will be billed. Known tokenizer families use exact native
|
|
47
|
+
* counts; unknown models keep the fast byte estimate (or o200k when
|
|
48
|
+
* `PI_TOKENIZER_ACCURATE=1`).
|
|
49
|
+
*/
|
|
50
|
+
export declare class Tokenizer {
|
|
51
|
+
#private;
|
|
52
|
+
constructor(model?: Pick<Model, "tokenizer"> | null);
|
|
53
|
+
get encoding(): Encoding | null;
|
|
54
|
+
countTokens(text: string | string[], mode?: TokenCountMode): number;
|
|
55
|
+
/**
|
|
56
|
+
* Cheap-first budget probe — the way to ask "does this fit in `budget`
|
|
57
|
+
* tokens?" without tokenizing the world.
|
|
58
|
+
*
|
|
59
|
+
* Byte length is a hard upper bound on token count (every token consumes at
|
|
60
|
+
* least one input byte), so text whose raw bytes already fit the budget
|
|
61
|
+
* cannot possibly exceed it — that verdict is returned without tokenizing at
|
|
62
|
+
* all. Only text that busts the bound is ambiguous, and only that case pays
|
|
63
|
+
* for the exact count. Since the bound overshoots ~4x on ordinary prose, the
|
|
64
|
+
* common "comfortably under budget" answer is free.
|
|
65
|
+
*/
|
|
66
|
+
checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck;
|
|
67
|
+
/**
|
|
68
|
+
* Token estimate for one message under this tokenizer's encoding.
|
|
69
|
+
*
|
|
70
|
+
* Settled historical messages are counted once and reused until an owner
|
|
71
|
+
* (prune/shake/strip-images) calls `invalidateMessageCache`; streaming
|
|
72
|
+
* assistants bypass the memo entirely (see the message-cache settle-gate
|
|
73
|
+
* invariant). Image blocks charge a fixed per-image estimate.
|
|
74
|
+
*/
|
|
75
|
+
countMessage(message: AgentMessage, options?: MessageCountOptions): number;
|
|
76
|
+
/** Sum of {@link countMessage} over `messages`. */
|
|
77
|
+
countMessages(messages: readonly AgentMessage[], options?: MessageCountOptions): number;
|
|
78
|
+
}
|
package/dist/types/types.d.ts
CHANGED
|
@@ -162,7 +162,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
|
|
|
162
162
|
* @example
|
|
163
163
|
* ```typescript
|
|
164
164
|
* transformContext: async (messages) => {
|
|
165
|
-
* if (
|
|
165
|
+
* if (agent.tokenizer.countMessages(messages) > MAX_TOKENS) {
|
|
166
166
|
* return pruneOldMessages(messages);
|
|
167
167
|
* }
|
|
168
168
|
* return messages;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-agent-core",
|
|
4
|
-
"version": "17.
|
|
4
|
+
"version": "17.4.0",
|
|
5
5
|
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": "Can Boluk",
|
|
@@ -35,16 +35,16 @@
|
|
|
35
35
|
"fmt": "biome format --write ."
|
|
36
36
|
},
|
|
37
37
|
"dependencies": {
|
|
38
|
-
"@oh-my-pi/pi-ai": "17.
|
|
39
|
-
"@oh-my-pi/pi-catalog": "17.
|
|
40
|
-
"@oh-my-pi/pi-natives": "17.
|
|
41
|
-
"@oh-my-pi/pi-utils": "17.
|
|
42
|
-
"@oh-my-pi/pi-wire": "17.
|
|
43
|
-
"@oh-my-pi/snapcompact": "17.
|
|
38
|
+
"@oh-my-pi/pi-ai": "17.4.0",
|
|
39
|
+
"@oh-my-pi/pi-catalog": "17.4.0",
|
|
40
|
+
"@oh-my-pi/pi-natives": "17.4.0",
|
|
41
|
+
"@oh-my-pi/pi-utils": "17.4.0",
|
|
42
|
+
"@oh-my-pi/pi-wire": "17.4.0",
|
|
43
|
+
"@oh-my-pi/snapcompact": "17.4.0",
|
|
44
44
|
"@opentelemetry/api": "^1.9.1"
|
|
45
45
|
},
|
|
46
46
|
"devDependencies": {
|
|
47
|
-
"@oh-my-pi/omptype": "17.
|
|
47
|
+
"@oh-my-pi/omptype": "17.4.0",
|
|
48
48
|
"@opentelemetry/context-async-hooks": "^2.9.0",
|
|
49
49
|
"@opentelemetry/sdk-trace-base": "^2.9.0",
|
|
50
50
|
"@types/bun": "^1.3.14"
|
package/src/agent.ts
CHANGED
|
@@ -37,6 +37,7 @@ import {
|
|
|
37
37
|
} from "./agent-loop";
|
|
38
38
|
import type { AppendOnlyContextManager } from "./append-only-context";
|
|
39
39
|
import { isProviderRefusalMessage } from "./replay-policy";
|
|
40
|
+
import { Tokenizer, tokenizerEncodingForModel } from "./tokenizer";
|
|
40
41
|
import type {
|
|
41
42
|
AgentBeforeModelCall,
|
|
42
43
|
AgentContext,
|
|
@@ -363,7 +364,7 @@ export class Agent {
|
|
|
363
364
|
pendingToolCalls: new Set<string>(),
|
|
364
365
|
error: undefined,
|
|
365
366
|
};
|
|
366
|
-
|
|
367
|
+
#tokenizer = new Tokenizer(this.#state.model);
|
|
367
368
|
#listeners = new Set<(e: AgentEvent) => void>();
|
|
368
369
|
#abortController?: AbortController;
|
|
369
370
|
#convertToLlm: (messages: AgentMessage[]) => Message[] | Promise<Message[]>;
|
|
@@ -459,6 +460,7 @@ export class Agent {
|
|
|
459
460
|
if (opts.initialState?.messages) this.#state.messages = opts.initialState.messages.slice();
|
|
460
461
|
if (opts.initialState?.pendingToolCalls)
|
|
461
462
|
this.#state.pendingToolCalls = new Set(opts.initialState.pendingToolCalls);
|
|
463
|
+
this.#syncTokenizer(this.#state.model);
|
|
462
464
|
this.#convertToLlm = opts.convertToLlm || defaultConvertToLlm;
|
|
463
465
|
this.#transformContext = opts.transformContext;
|
|
464
466
|
this.#steeringMode = opts.steeringMode || "one-at-a-time";
|
|
@@ -720,11 +722,29 @@ export class Agent {
|
|
|
720
722
|
set maxRetryDelayMs(value: number | undefined) {
|
|
721
723
|
this.#maxRetryDelayMs = value;
|
|
722
724
|
}
|
|
723
|
-
|
|
724
725
|
get state(): AgentState {
|
|
725
726
|
return this.#state;
|
|
726
727
|
}
|
|
727
728
|
|
|
729
|
+
/**
|
|
730
|
+
* Tokenizer for the active model. The instance is replaced whenever the
|
|
731
|
+
* active model's encoding changes (see {@link setModel}), so callers must
|
|
732
|
+
* not cache it across model switches.
|
|
733
|
+
*/
|
|
734
|
+
get tokenizer(): Tokenizer {
|
|
735
|
+
return this.#tokenizer;
|
|
736
|
+
}
|
|
737
|
+
|
|
738
|
+
/**
|
|
739
|
+
* Swap the tokenizer only when the encoding actually changes, so the warm
|
|
740
|
+
* per-message memo survives same-encoding model switches.
|
|
741
|
+
*/
|
|
742
|
+
#syncTokenizer(model: Model | null | undefined): void {
|
|
743
|
+
if (tokenizerEncodingForModel(model) !== this.#tokenizer.encoding) {
|
|
744
|
+
this.#tokenizer = new Tokenizer(model);
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
|
|
728
748
|
get appendOnlyContext(): AppendOnlyContextManager | undefined {
|
|
729
749
|
return this.#appendOnlyContext;
|
|
730
750
|
}
|
|
@@ -902,8 +922,9 @@ export class Agent {
|
|
|
902
922
|
this.#state.systemPrompt = typeof v === "string" ? [v] : v;
|
|
903
923
|
}
|
|
904
924
|
|
|
905
|
-
setModel(
|
|
906
|
-
this.#state.model =
|
|
925
|
+
setModel(model: Model) {
|
|
926
|
+
this.#state.model = model;
|
|
927
|
+
this.#syncTokenizer(model);
|
|
907
928
|
}
|
|
908
929
|
|
|
909
930
|
setThinkingLevel(l: Effort | undefined) {
|