@oh-my-pi/pi-agent-core 17.3.8 → 17.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/LICENSE +23 -0
- package/THIRD-PARTY-NOTICES.txt +22909 -0
- package/dist/types/agent.d.ts +8 -1
- package/dist/types/compaction/branch-summarization.d.ts +2 -1
- package/dist/types/compaction/compaction.d.ts +24 -17
- package/dist/types/compaction/index.d.ts +1 -0
- package/dist/types/compaction/message-cache.d.ts +6 -4
- package/dist/types/compaction/messages.d.ts +17 -1
- package/dist/types/compaction/openai.d.ts +2 -1
- package/dist/types/compaction/pruning.d.ts +3 -2
- package/dist/types/compaction/shake.d.ts +2 -1
- package/dist/types/compaction/transcript-tokens.d.ts +76 -0
- package/dist/types/tokenizer.d.ts +78 -2
- package/dist/types/types.d.ts +1 -1
- package/package.json +11 -9
- package/src/agent.ts +25 -4
- package/src/compaction/branch-summarization.ts +15 -8
- package/src/compaction/compaction-v2-streaming.ts +2 -0
- package/src/compaction/compaction.ts +43 -152
- package/src/compaction/index.ts +1 -0
- package/src/compaction/message-cache.ts +33 -34
- package/src/compaction/messages.ts +21 -5
- package/src/compaction/openai.ts +48 -18
- package/src/compaction/pruning.ts +25 -13
- package/src/compaction/shake.ts +18 -16
- package/src/compaction/transcript-tokens.ts +111 -0
- package/src/tokenizer.ts +273 -18
- package/src/types.ts +1 -1
package/dist/types/agent.d.ts
CHANGED
|
@@ -2,6 +2,7 @@ import { type ApiKey, type AssistantMessage, type AssistantMessageEvent, type Co
|
|
|
2
2
|
import type { Dialect } from "@oh-my-pi/pi-ai/dialect";
|
|
3
3
|
import type { HarmonyAuditEvent } from "@oh-my-pi/pi-ai/utils/harmony-leak";
|
|
4
4
|
import type { AppendOnlyContextManager } from "./append-only-context.js";
|
|
5
|
+
import { Tokenizer } from "./tokenizer.js";
|
|
5
6
|
import type { AgentBeforeModelCall, AgentEvent, AgentLoopConfig, AgentMessage, AgentState, AgentTool, AgentToolContext, AgentTurnEndContext, AsideMessage, StreamFn, ToolCallContext, ToolChoiceDirective } from "./types.js";
|
|
6
7
|
export declare class AgentBusyError extends Error {
|
|
7
8
|
constructor(message?: string);
|
|
@@ -354,6 +355,12 @@ export declare class Agent {
|
|
|
354
355
|
*/
|
|
355
356
|
set maxRetryDelayMs(value: number | undefined);
|
|
356
357
|
get state(): AgentState;
|
|
358
|
+
/**
|
|
359
|
+
* Tokenizer for the active model. The instance is replaced whenever the
|
|
360
|
+
* active model's encoding changes (see {@link setModel}), so callers must
|
|
361
|
+
* not cache it across model switches.
|
|
362
|
+
*/
|
|
363
|
+
get tokenizer(): Tokenizer;
|
|
357
364
|
get appendOnlyContext(): AppendOnlyContextManager | undefined;
|
|
358
365
|
setAppendOnlyContext(manager?: AppendOnlyContextManager): void;
|
|
359
366
|
/**
|
|
@@ -401,7 +408,7 @@ export declare class Agent {
|
|
|
401
408
|
setAsideMessageProvider(fn: (() => AsideMessage[] | Promise<AsideMessage[]>) | undefined): void;
|
|
402
409
|
emitExternalEvent(event: AgentEvent): void;
|
|
403
410
|
setSystemPrompt(v: string[] | string): void;
|
|
404
|
-
setModel(
|
|
411
|
+
setModel(model: Model): void;
|
|
405
412
|
setThinkingLevel(l: Effort | undefined): void;
|
|
406
413
|
setDisableReasoning(disabled: boolean): void;
|
|
407
414
|
setSteeringMode(mode: "all" | "one-at-a-time"): void;
|
|
@@ -6,6 +6,7 @@
|
|
|
6
6
|
*/
|
|
7
7
|
import type { Api, ApiKey, AssistantMessage, Context, Model, SimpleStreamOptions } from "@oh-my-pi/pi-ai";
|
|
8
8
|
import { type AgentTelemetry } from "../telemetry.js";
|
|
9
|
+
import { Tokenizer } from "../tokenizer.js";
|
|
9
10
|
import type { AgentMessage } from "../types.js";
|
|
10
11
|
import type { ReadonlySessionManager, SessionEntry } from "./entries.js";
|
|
11
12
|
import { type ConvertToLlm } from "./messages.js";
|
|
@@ -91,7 +92,7 @@ export declare function collectEntriesForBranchSummary(session: ReadonlySessionM
|
|
|
91
92
|
* @param entries - Entries in chronological order
|
|
92
93
|
* @param tokenBudget - Maximum tokens to include (0 = no limit)
|
|
93
94
|
*/
|
|
94
|
-
export declare function prepareBranchEntries(entries: SessionEntry[], tokenBudget?: number): BranchPreparation;
|
|
95
|
+
export declare function prepareBranchEntries(entries: SessionEntry[], tokenizer: Tokenizer, tokenBudget?: number): BranchPreparation;
|
|
95
96
|
/**
|
|
96
97
|
* Generate a summary of abandoned branch entries.
|
|
97
98
|
*
|
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
import { type Api, type ApiKey, type AssistantMessage, type CodexCompactionContext, type Context, type FetchImpl, type MessageAttribution, type Model, type OneshotRetryOptions, type ProviderSessionState, type SimpleStreamOptions, type Tool, type Usage } from "@oh-my-pi/pi-ai";
|
|
8
8
|
import { type AgentTelemetry } from "../telemetry.js";
|
|
9
9
|
import { ThinkingLevel } from "../thinking.js";
|
|
10
|
+
import { Tokenizer } from "../tokenizer.js";
|
|
10
11
|
import type { AgentMessage } from "../types.js";
|
|
11
12
|
import type { SessionEntry } from "./entries.js";
|
|
12
13
|
import { type ConvertToLlm } from "./messages.js";
|
|
@@ -119,21 +120,6 @@ export declare function shouldCompact(contextTokens: number, contextWindow: numb
|
|
|
119
120
|
*/
|
|
120
121
|
export declare function compactionContextTokens(providerContextTokens: number, storedConversationEstimate: number): number;
|
|
121
122
|
export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings): number;
|
|
122
|
-
/**
|
|
123
|
-
* Estimate token count for a message using cl100k_base via the native
|
|
124
|
-
* tokenizer. This is not Claude's first-party tokenizer (Anthropic doesn't
|
|
125
|
-
* publish one) but is within ~5–10% across English/code text.
|
|
126
|
-
*
|
|
127
|
-
* `excludeEncryptedReasoning` drops opaque provider reasoning payloads
|
|
128
|
-
* (`thinkingSignature`, `redactedThinking`) from the estimate. Those are billed
|
|
129
|
-
* by the provider on replay, so the default counts them — but their *local*
|
|
130
|
-
* byte size can diverge wildly from what the provider charges, so the
|
|
131
|
-
* compaction floor (which only needs the reliably-countable, on-wire-compressible
|
|
132
|
-
* content) excludes them to avoid false triggers on thinking-heavy turns.
|
|
133
|
-
*/
|
|
134
|
-
export declare function estimateTokens(message: AgentMessage, options?: {
|
|
135
|
-
excludeEncryptedReasoning?: boolean;
|
|
136
|
-
}): number;
|
|
137
123
|
/**
|
|
138
124
|
* Find the user message (or bashExecution) that starts the turn containing the given entry index.
|
|
139
125
|
* Returns -1 if no turn start found before the index.
|
|
@@ -164,7 +150,7 @@ export interface CutPointResult {
|
|
|
164
150
|
*
|
|
165
151
|
* Only considers entries between `startIndex` and `endIndex` (exclusive).
|
|
166
152
|
*/
|
|
167
|
-
export declare function findCutPoint(entries: SessionEntry[], startIndex: number, endIndex: number, keepRecentTokens: number): CutPointResult;
|
|
153
|
+
export declare function findCutPoint(entries: SessionEntry[], tokenizer: Tokenizer, startIndex: number, endIndex: number, keepRecentTokens: number): CutPointResult;
|
|
168
154
|
export declare const AUTO_HANDOFF_THRESHOLD_FOCUS: string;
|
|
169
155
|
/**
|
|
170
156
|
* Generate a summary of the conversation using the LLM.
|
|
@@ -310,6 +296,22 @@ export interface CompactionPreparation {
|
|
|
310
296
|
/** Compaction settions from settings.jsonl */
|
|
311
297
|
settings: CompactionSettings;
|
|
312
298
|
}
|
|
299
|
+
/**
|
|
300
|
+
* Whether a prior remote compaction's provider-native replay can still be read
|
|
301
|
+
* by the active model — the model that assembles the request context on every
|
|
302
|
+
* turn. A local compaction (no remote preserve) always can: it holds a real
|
|
303
|
+
* textual summary. A remote compaction (V2 or V1) only can when the active model
|
|
304
|
+
* shares the blob's provider AND remote replay is still enabled; otherwise the
|
|
305
|
+
* active model's encoder drops the payload (see `getOpenAIResponsesHistoryPayload`)
|
|
306
|
+
* and only the opaque placeholder summary survives, so the caller must re-expand
|
|
307
|
+
* the originals into a portable local summary rather than strand that history.
|
|
308
|
+
*
|
|
309
|
+
* Judged against the ACTIVE model, not the compaction candidate set: a role
|
|
310
|
+
* model (e.g. `modelRoles.smol`) that still maps to the blob's provider does not
|
|
311
|
+
* let the active model replay it, so keying reuse on "any candidate shares the
|
|
312
|
+
* provider" left a provider-switched session permanently context-less (#6343).
|
|
313
|
+
*/
|
|
314
|
+
export declare function remotePreserveReusable(preserveData: Record<string, unknown> | undefined, activeModel: Model, settings: CompactionSettings): boolean;
|
|
313
315
|
/**
|
|
314
316
|
* Index of the newest compaction entry the active model can actually read, or
|
|
315
317
|
* `-1` when none can.
|
|
@@ -323,7 +325,12 @@ export interface CompactionPreparation {
|
|
|
323
325
|
* entries that no summary covers.
|
|
324
326
|
*/
|
|
325
327
|
export declare function findReadableCompactionIndex(pathEntries: SessionEntry[], settings: CompactionSettings, activeModel?: Model): number;
|
|
326
|
-
|
|
328
|
+
/**
|
|
329
|
+
* Pass the caller's warm `tokenizer` (the Agent's for the active model) so the
|
|
330
|
+
* full-branch estimate walk hits its memo; the cold default is for one-shot
|
|
331
|
+
* callers that have no live agent.
|
|
332
|
+
*/
|
|
333
|
+
export declare function prepareCompaction(pathEntries: SessionEntry[], settings: CompactionSettings, activeModel?: Model, tokenizer?: Tokenizer): CompactionPreparation | undefined;
|
|
327
334
|
/**
|
|
328
335
|
* Generate summaries for compaction using prepared data.
|
|
329
336
|
* Returns CompactionResult - SessionManager adds id/parentId when saving.
|
|
@@ -5,16 +5,18 @@ import type { AgentMessage } from "../types.js";
|
|
|
5
5
|
* unregister function. The coding-agent `convertToLlm` memo registers here.
|
|
6
6
|
*/
|
|
7
7
|
export declare function registerMessageCacheInvalidator(invalidate: (message: AgentMessage) => void): () => void;
|
|
8
|
+
/**
|
|
9
|
+
* Current estimate version of `message` (0 until first invalidation). A
|
|
10
|
+
* `Tokenizer` memo entry stamped with an older version is stale and must be
|
|
11
|
+
* recounted.
|
|
12
|
+
*/
|
|
13
|
+
export declare function messageEstimateVersion(message: AgentMessage): number;
|
|
8
14
|
/**
|
|
9
15
|
* True when this message's estimate is safe to cache by identity. Non-assistants
|
|
10
16
|
* are immutable once appended; assistants are cached only once settled (see the
|
|
11
17
|
* settle-gate invariant above).
|
|
12
18
|
*/
|
|
13
19
|
export declare function isEstimateCacheable(message: AgentMessage): boolean;
|
|
14
|
-
/** Read a cached estimate for the given option split, or `undefined` on miss. */
|
|
15
|
-
export declare function readEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean): number | undefined;
|
|
16
|
-
/** Store an estimate for the given option split. */
|
|
17
|
-
export declare function writeEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean, value: number): void;
|
|
18
20
|
/**
|
|
19
21
|
* Drop every cached derivation of `message` after an in-place rewrite. Owners of
|
|
20
22
|
* mutation (prune, shake, strip-images) call this at the mutation seam so the
|
|
@@ -32,6 +32,10 @@ export interface CompactionSummaryMessage {
|
|
|
32
32
|
summary: string;
|
|
33
33
|
shortSummary?: string;
|
|
34
34
|
tokensBefore: number;
|
|
35
|
+
/** Estimated context tokens after the rewrite (display metadata). */
|
|
36
|
+
tokensAfter?: number;
|
|
37
|
+
/** Harness compaction method that produced this summary (display metadata). */
|
|
38
|
+
method?: string;
|
|
35
39
|
providerPayload?: ProviderPayload;
|
|
36
40
|
/** Runtime-only ordered archive blocks for snapcompact: old text region,
|
|
37
41
|
* imaged middle, then new text region. When present, `summary` is already
|
|
@@ -56,7 +60,19 @@ export type ConvertToLlm = (messages: AgentMessage[]) => Message[];
|
|
|
56
60
|
export declare function renderBranchSummaryContext(summary: string): string;
|
|
57
61
|
export declare function renderCompactionSummaryContext(summary: string): string;
|
|
58
62
|
export declare function createBranchSummaryMessage(summary: string, fromId: string, timestamp: string): BranchSummaryMessage;
|
|
59
|
-
|
|
63
|
+
/** Optional metadata for {@link createCompactionSummaryMessage}. */
|
|
64
|
+
export interface CompactionSummaryMessageOptions {
|
|
65
|
+
shortSummary?: string;
|
|
66
|
+
providerPayload?: ProviderPayload;
|
|
67
|
+
images?: ImageContent[];
|
|
68
|
+
blocks?: (TextContent | ImageContent)[];
|
|
69
|
+
warning?: string;
|
|
70
|
+
/** Harness compaction method that produced this summary (e.g. "remote", "soft", "handoff"). */
|
|
71
|
+
method?: string;
|
|
72
|
+
/** Estimated context tokens after the rewrite, for display alongside `tokensBefore`. */
|
|
73
|
+
tokensAfter?: number;
|
|
74
|
+
}
|
|
75
|
+
export declare function createCompactionSummaryMessage(summary: string, tokensBefore: number, timestamp: string, options?: CompactionSummaryMessageOptions): CompactionSummaryMessage;
|
|
60
76
|
export declare function createCustomMessage(customType: string, content: string | (TextContent | ImageContent)[], display: boolean, details: unknown | undefined, timestamp: string, attribution?: MessageAttribution): CustomMessage;
|
|
61
77
|
/**
|
|
62
78
|
* Transform a single core-domain agent message to its LLM form; `undefined`
|
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
* with `{ summary, shortSummary? }`.
|
|
16
16
|
*/
|
|
17
17
|
import type { CodexCompactionContext, FetchImpl, Message, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
|
|
18
|
+
import { Tokenizer } from "../tokenizer.js";
|
|
18
19
|
export * from "./compaction-v2-streaming.js";
|
|
19
20
|
export declare const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
|
|
20
21
|
/**
|
|
@@ -39,7 +40,7 @@ export interface TrimRemoteCompactionInputResult {
|
|
|
39
40
|
* outputs keeps call/result pairing and all earlier assistant/reasoning history,
|
|
40
41
|
* matching Codex's recovery path for oversized tool turns.
|
|
41
42
|
*/
|
|
42
|
-
export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
|
|
43
|
+
export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, tokenizer: Tokenizer, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
|
|
43
44
|
export type OpenAiRemoteCompactionItem = {
|
|
44
45
|
type: "compaction" | "compaction_summary";
|
|
45
46
|
encrypted_content?: string;
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Tool output pruning utilities for compaction.
|
|
3
3
|
*/
|
|
4
|
+
import type { Tokenizer } from "../tokenizer.js";
|
|
4
5
|
import type { SessionEntry } from "./entries.js";
|
|
5
6
|
import { type ProtectedToolMatcher } from "./tool-protection.js";
|
|
6
7
|
export interface PruneConfig {
|
|
@@ -90,8 +91,8 @@ export interface SupersedePruneConfig {
|
|
|
90
91
|
* the provider cache is cold anyway (then all still-sent candidates flush).
|
|
91
92
|
* Never mutates entries before `keepBoundaryId` (summarized away — not sent).
|
|
92
93
|
*/
|
|
93
|
-
export declare function pruneSupersededToolResults(entries: SessionEntry[], config: SupersedePruneConfig): PruneResult;
|
|
94
|
-
export declare function pruneToolOutputs(entries: SessionEntry[], config?: PruneConfig): PruneResult;
|
|
94
|
+
export declare function pruneSupersededToolResults(entries: SessionEntry[], tokenizer: Tokenizer, config: SupersedePruneConfig): PruneResult;
|
|
95
|
+
export declare function pruneToolOutputs(entries: SessionEntry[], tokenizer: Tokenizer, config?: PruneConfig): PruneResult;
|
|
95
96
|
/**
|
|
96
97
|
* Supersede key for the `read` tool: the file path with the trailing line/raw
|
|
97
98
|
* selector stripped (the read tool's own splitter grammar via
|
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
*
|
|
10
10
|
* Layering mirrors `pruning.ts`: no I/O here.
|
|
11
11
|
*/
|
|
12
|
+
import type { Tokenizer } from "../tokenizer.js";
|
|
12
13
|
import type { CustomMessageEntry, SessionEntry, SessionMessageEntry } from "./entries.js";
|
|
13
14
|
import { type ProtectedToolMatcher } from "./tool-protection.js";
|
|
14
15
|
export interface ShakeConfig {
|
|
@@ -77,7 +78,7 @@ export type ShakeRegion = ToolResultShakeRegion | BlockShakeRegion;
|
|
|
77
78
|
* and regions never span a message boundary. When the combined estimated
|
|
78
79
|
* savings is below `minSavings`, returns `[]` (no-op).
|
|
79
80
|
*/
|
|
80
|
-
export declare function collectShakeRegions(entries: SessionEntry[], config: ShakeConfig): ShakeRegion[];
|
|
81
|
+
export declare function collectShakeRegions(entries: SessionEntry[], tokenizer: Tokenizer, config: ShakeConfig): ShakeRegion[];
|
|
81
82
|
/**
|
|
82
83
|
* Pure mutation: replace a single region's content in place.
|
|
83
84
|
*
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Provider-anchored transcript token accounting.
|
|
3
|
+
*
|
|
4
|
+
* Local tokenization is the expensive way to answer "how big is this
|
|
5
|
+
* conversation?" — and usually the wrong one, because the provider already
|
|
6
|
+
* answered it. Every settled assistant turn carries `usage` covering the exact
|
|
7
|
+
* prompt it was sent: the system prompt, the tool schemas, and every message up
|
|
8
|
+
* to and including itself. The only genuinely unaccounted-for text is the tail
|
|
9
|
+
* appended *after* that turn.
|
|
10
|
+
*
|
|
11
|
+
* These helpers locate the newest trustworthy usage report and tokenize only
|
|
12
|
+
* that tail, so a long session pays counting proportional to one turn instead
|
|
13
|
+
* of to the whole transcript, every turn.
|
|
14
|
+
*
|
|
15
|
+
* Trust rules for an anchor (mirroring the provider contract):
|
|
16
|
+
* - Assistant role only — nothing else carries `usage`.
|
|
17
|
+
* - Not `aborted` / `error`: those turns report partial or zero usage.
|
|
18
|
+
* - `hasContextTokenUsage(usage)`: the report must carry usable context numbers.
|
|
19
|
+
*/
|
|
20
|
+
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
|
|
21
|
+
import type { Tokenizer } from "../tokenizer.js";
|
|
22
|
+
import type { AgentMessage } from "../types.js";
|
|
23
|
+
/** A provider usage report that accounts for a prefix of the transcript. */
|
|
24
|
+
export interface TranscriptUsageAnchor {
|
|
25
|
+
/** Index in the scanned array; messages at or before it are provider-accounted. */
|
|
26
|
+
index: number;
|
|
27
|
+
/** The anchoring assistant turn. */
|
|
28
|
+
message: AssistantMessage;
|
|
29
|
+
/** Conversation tokens the provider reported for that prompt. */
|
|
30
|
+
tokens: number;
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* Whether this message's provider usage may anchor transcript accounting.
|
|
34
|
+
*
|
|
35
|
+
* The single home for the trust rules — every anchor scan MUST route through
|
|
36
|
+
* it so a stale-usage rule can never drift between the transcript walkers and
|
|
37
|
+
* the session-entry walkers.
|
|
38
|
+
*/
|
|
39
|
+
export declare function isTranscriptUsageAnchor(message: AgentMessage): message is AssistantMessage;
|
|
40
|
+
/**
|
|
41
|
+
* Newest assistant turn in `messages[fromIndex..]` whose usage can anchor the
|
|
42
|
+
* transcript, or `undefined` when none qualifies (fresh context, or every
|
|
43
|
+
* recent turn aborted/errored).
|
|
44
|
+
*
|
|
45
|
+
* `fromIndex` excludes turns whose usage is stale — anything a compaction
|
|
46
|
+
* summarized away describes a prompt that is no longer sent.
|
|
47
|
+
*/
|
|
48
|
+
export declare function findTranscriptUsageAnchor(messages: readonly AgentMessage[], fromIndex?: number): TranscriptUsageAnchor | undefined;
|
|
49
|
+
/** Options for {@link estimateTranscriptTokens}. */
|
|
50
|
+
export interface TranscriptTokenOptions {
|
|
51
|
+
/**
|
|
52
|
+
* Gates the anchor search only: usage at or before this index is stale (a
|
|
53
|
+
* compaction rewrote the prompt it describes) and must not anchor. Content
|
|
54
|
+
* accounting is governed separately by {@link countFromIndex}.
|
|
55
|
+
*/
|
|
56
|
+
anchorFromIndex?: number;
|
|
57
|
+
/**
|
|
58
|
+
* First message whose content is counted locally when no anchor is found.
|
|
59
|
+
* Defaults to 0 (count the whole transcript), which is what a floor
|
|
60
|
+
* estimate wants; pass the compaction boundary to skip summarized-away
|
|
61
|
+
* messages entirely.
|
|
62
|
+
*/
|
|
63
|
+
countFromIndex?: number;
|
|
64
|
+
/** Forwarded to {@link Tokenizer.countMessage} for every locally counted message. */
|
|
65
|
+
excludeEncryptedReasoning?: boolean;
|
|
66
|
+
}
|
|
67
|
+
/**
|
|
68
|
+
* Conversation tokens for `messages`: the provider's own report for everything
|
|
69
|
+
* it already covers, plus a local count of only the unaccounted-for tail.
|
|
70
|
+
*
|
|
71
|
+
* An anchored result already includes the non-message prefix (system prompt +
|
|
72
|
+
* tool schemas) because the provider charged it; an unanchored result is a
|
|
73
|
+
* message-only sum. Callers that add non-message tokens on top MUST branch on
|
|
74
|
+
* {@link findTranscriptUsageAnchor} rather than assuming one shape.
|
|
75
|
+
*/
|
|
76
|
+
export declare function estimateTranscriptTokens(messages: readonly AgentMessage[], tokenizer: Tokenizer, options?: TranscriptTokenOptions): number;
|
|
@@ -1,2 +1,78 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
1
|
+
import type { Model } from "@oh-my-pi/pi-ai";
|
|
2
|
+
import { Encoding } from "@oh-my-pi/pi-natives";
|
|
3
|
+
import type { AgentMessage } from "./types.js";
|
|
4
|
+
/** Maps the catalog-resolved tokenizer family to its native implementation. */
|
|
5
|
+
export declare function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): Encoding | null;
|
|
6
|
+
/**
|
|
7
|
+
* `strict` always pays for an exact native count (the catalog-resolved
|
|
8
|
+
* tokenizer when known, o200k_base otherwise). `approximate` and
|
|
9
|
+
* `upperbound` prefer the same exact count for known tokenizer families or
|
|
10
|
+
* when `PI_TOKENIZER_ACCURATE=1` is set; otherwise they use a cheap heuristic:
|
|
11
|
+
* `approximate` a bytes/4 guess, `upperbound` the raw byte length (never
|
|
12
|
+
* undercounts).
|
|
13
|
+
*/
|
|
14
|
+
export type TokenCountMode = "strict" | "approximate" | "upperbound";
|
|
15
|
+
/** Options for {@link Tokenizer.countMessage} / {@link Tokenizer.countMessages}. */
|
|
16
|
+
export interface MessageCountOptions {
|
|
17
|
+
/**
|
|
18
|
+
* Drop opaque provider reasoning payloads (`thinkingSignature`,
|
|
19
|
+
* `redactedThinking`, native server-tool blocks) from the estimate. Those
|
|
20
|
+
* are billed by the provider on replay, so the default counts them — but
|
|
21
|
+
* their *local* byte size can diverge wildly from what the provider
|
|
22
|
+
* charges, so the compaction floor (which only needs the reliably-countable,
|
|
23
|
+
* on-wire-compressible content) excludes them to avoid false triggers on
|
|
24
|
+
* thinking-heavy turns.
|
|
25
|
+
*/
|
|
26
|
+
excludeEncryptedReasoning?: boolean;
|
|
27
|
+
}
|
|
28
|
+
/** Verdict from {@link Tokenizer.checkTokenBudget}. */
|
|
29
|
+
export interface TokenBudgetCheck {
|
|
30
|
+
/** Whether the text fits the budget. */
|
|
31
|
+
fits: boolean;
|
|
32
|
+
/**
|
|
33
|
+
* Token count behind the verdict: the exact native count when `exact` is
|
|
34
|
+
* set, otherwise the cheap byte upper bound (which already fit, so it is
|
|
35
|
+
* only an over-estimate of a count known to be under budget).
|
|
36
|
+
*/
|
|
37
|
+
tokens: number;
|
|
38
|
+
/** Whether the exact tokenizer had to run because the cheap bound busted. */
|
|
39
|
+
exact: boolean;
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Model-aware local token counter. Immutable: the catalog-resolved encoding
|
|
43
|
+
* is fixed at construction, so a cached count can never straddle two
|
|
44
|
+
* encodings. An `Agent` owns one for its active model (swapping the instance
|
|
45
|
+
* when the model's encoding changes); one-shot flows construct their own for
|
|
46
|
+
* the model that will be billed. Known tokenizer families use exact native
|
|
47
|
+
* counts; unknown models keep the fast byte estimate (or o200k when
|
|
48
|
+
* `PI_TOKENIZER_ACCURATE=1`).
|
|
49
|
+
*/
|
|
50
|
+
export declare class Tokenizer {
|
|
51
|
+
#private;
|
|
52
|
+
constructor(model?: Pick<Model, "tokenizer"> | null);
|
|
53
|
+
get encoding(): Encoding | null;
|
|
54
|
+
countTokens(text: string | string[], mode?: TokenCountMode): number;
|
|
55
|
+
/**
|
|
56
|
+
* Cheap-first budget probe — the way to ask "does this fit in `budget`
|
|
57
|
+
* tokens?" without tokenizing the world.
|
|
58
|
+
*
|
|
59
|
+
* Byte length is a hard upper bound on token count (every token consumes at
|
|
60
|
+
* least one input byte), so text whose raw bytes already fit the budget
|
|
61
|
+
* cannot possibly exceed it — that verdict is returned without tokenizing at
|
|
62
|
+
* all. Only text that busts the bound is ambiguous, and only that case pays
|
|
63
|
+
* for the exact count. Since the bound overshoots ~4x on ordinary prose, the
|
|
64
|
+
* common "comfortably under budget" answer is free.
|
|
65
|
+
*/
|
|
66
|
+
checkTokenBudget(text: string | string[], budget: number): TokenBudgetCheck;
|
|
67
|
+
/**
|
|
68
|
+
* Token estimate for one message under this tokenizer's encoding.
|
|
69
|
+
*
|
|
70
|
+
* Settled historical messages are counted once and reused until an owner
|
|
71
|
+
* (prune/shake/strip-images) calls `invalidateMessageCache`; streaming
|
|
72
|
+
* assistants bypass the memo entirely (see the message-cache settle-gate
|
|
73
|
+
* invariant). Image blocks charge a fixed per-image estimate.
|
|
74
|
+
*/
|
|
75
|
+
countMessage(message: AgentMessage, options?: MessageCountOptions): number;
|
|
76
|
+
/** Sum of {@link countMessage} over `messages`. */
|
|
77
|
+
countMessages(messages: readonly AgentMessage[], options?: MessageCountOptions): number;
|
|
78
|
+
}
|
package/dist/types/types.d.ts
CHANGED
|
@@ -162,7 +162,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
|
|
|
162
162
|
* @example
|
|
163
163
|
* ```typescript
|
|
164
164
|
* transformContext: async (messages) => {
|
|
165
|
-
* if (
|
|
165
|
+
* if (agent.tokenizer.countMessages(messages) > MAX_TOKENS) {
|
|
166
166
|
* return pruneOldMessages(messages);
|
|
167
167
|
* }
|
|
168
168
|
* return messages;
|
package/package.json
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-agent-core",
|
|
4
|
-
"version": "17.
|
|
4
|
+
"version": "17.4.1",
|
|
5
5
|
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
|
-
"author": "
|
|
7
|
+
"author": "Stencil Labs, Inc.",
|
|
8
8
|
"contributors": [
|
|
9
9
|
"Mario Zechner"
|
|
10
10
|
],
|
|
@@ -35,16 +35,16 @@
|
|
|
35
35
|
"fmt": "biome format --write ."
|
|
36
36
|
},
|
|
37
37
|
"dependencies": {
|
|
38
|
-
"@oh-my-pi/pi-ai": "17.
|
|
39
|
-
"@oh-my-pi/pi-catalog": "17.
|
|
40
|
-
"@oh-my-pi/pi-natives": "17.
|
|
41
|
-
"@oh-my-pi/pi-utils": "17.
|
|
42
|
-
"@oh-my-pi/pi-wire": "17.
|
|
43
|
-
"@oh-my-pi/snapcompact": "17.
|
|
38
|
+
"@oh-my-pi/pi-ai": "17.4.1",
|
|
39
|
+
"@oh-my-pi/pi-catalog": "17.4.1",
|
|
40
|
+
"@oh-my-pi/pi-natives": "17.4.1",
|
|
41
|
+
"@oh-my-pi/pi-utils": "17.4.1",
|
|
42
|
+
"@oh-my-pi/pi-wire": "17.4.1",
|
|
43
|
+
"@oh-my-pi/snapcompact": "17.4.1",
|
|
44
44
|
"@opentelemetry/api": "^1.9.1"
|
|
45
45
|
},
|
|
46
46
|
"devDependencies": {
|
|
47
|
-
"@oh-my-pi/omptype": "17.
|
|
47
|
+
"@oh-my-pi/omptype": "17.4.1",
|
|
48
48
|
"@opentelemetry/context-async-hooks": "^2.9.0",
|
|
49
49
|
"@opentelemetry/sdk-trace-base": "^2.9.0",
|
|
50
50
|
"@types/bun": "^1.3.14"
|
|
@@ -56,6 +56,8 @@
|
|
|
56
56
|
"src",
|
|
57
57
|
"README.md",
|
|
58
58
|
"CHANGELOG.md",
|
|
59
|
+
"LICENSE",
|
|
60
|
+
"THIRD-PARTY-NOTICES.txt",
|
|
59
61
|
"dist/types"
|
|
60
62
|
],
|
|
61
63
|
"exports": {
|
package/src/agent.ts
CHANGED
|
@@ -37,6 +37,7 @@ import {
|
|
|
37
37
|
} from "./agent-loop";
|
|
38
38
|
import type { AppendOnlyContextManager } from "./append-only-context";
|
|
39
39
|
import { isProviderRefusalMessage } from "./replay-policy";
|
|
40
|
+
import { Tokenizer, tokenizerEncodingForModel } from "./tokenizer";
|
|
40
41
|
import type {
|
|
41
42
|
AgentBeforeModelCall,
|
|
42
43
|
AgentContext,
|
|
@@ -363,7 +364,7 @@ export class Agent {
|
|
|
363
364
|
pendingToolCalls: new Set<string>(),
|
|
364
365
|
error: undefined,
|
|
365
366
|
};
|
|
366
|
-
|
|
367
|
+
#tokenizer = new Tokenizer(this.#state.model);
|
|
367
368
|
#listeners = new Set<(e: AgentEvent) => void>();
|
|
368
369
|
#abortController?: AbortController;
|
|
369
370
|
#convertToLlm: (messages: AgentMessage[]) => Message[] | Promise<Message[]>;
|
|
@@ -459,6 +460,7 @@ export class Agent {
|
|
|
459
460
|
if (opts.initialState?.messages) this.#state.messages = opts.initialState.messages.slice();
|
|
460
461
|
if (opts.initialState?.pendingToolCalls)
|
|
461
462
|
this.#state.pendingToolCalls = new Set(opts.initialState.pendingToolCalls);
|
|
463
|
+
this.#syncTokenizer(this.#state.model);
|
|
462
464
|
this.#convertToLlm = opts.convertToLlm || defaultConvertToLlm;
|
|
463
465
|
this.#transformContext = opts.transformContext;
|
|
464
466
|
this.#steeringMode = opts.steeringMode || "one-at-a-time";
|
|
@@ -720,11 +722,29 @@ export class Agent {
|
|
|
720
722
|
set maxRetryDelayMs(value: number | undefined) {
|
|
721
723
|
this.#maxRetryDelayMs = value;
|
|
722
724
|
}
|
|
723
|
-
|
|
724
725
|
get state(): AgentState {
|
|
725
726
|
return this.#state;
|
|
726
727
|
}
|
|
727
728
|
|
|
729
|
+
/**
|
|
730
|
+
* Tokenizer for the active model. The instance is replaced whenever the
|
|
731
|
+
* active model's encoding changes (see {@link setModel}), so callers must
|
|
732
|
+
* not cache it across model switches.
|
|
733
|
+
*/
|
|
734
|
+
get tokenizer(): Tokenizer {
|
|
735
|
+
return this.#tokenizer;
|
|
736
|
+
}
|
|
737
|
+
|
|
738
|
+
/**
|
|
739
|
+
* Swap the tokenizer only when the encoding actually changes, so the warm
|
|
740
|
+
* per-message memo survives same-encoding model switches.
|
|
741
|
+
*/
|
|
742
|
+
#syncTokenizer(model: Model | null | undefined): void {
|
|
743
|
+
if (tokenizerEncodingForModel(model) !== this.#tokenizer.encoding) {
|
|
744
|
+
this.#tokenizer = new Tokenizer(model);
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
|
|
728
748
|
get appendOnlyContext(): AppendOnlyContextManager | undefined {
|
|
729
749
|
return this.#appendOnlyContext;
|
|
730
750
|
}
|
|
@@ -902,8 +922,9 @@ export class Agent {
|
|
|
902
922
|
this.#state.systemPrompt = typeof v === "string" ? [v] : v;
|
|
903
923
|
}
|
|
904
924
|
|
|
905
|
-
setModel(
|
|
906
|
-
this.#state.model =
|
|
925
|
+
setModel(model: Model) {
|
|
926
|
+
this.#state.model = model;
|
|
927
|
+
this.#syncTokenizer(model);
|
|
907
928
|
}
|
|
908
929
|
|
|
909
930
|
setThinkingLevel(l: Effort | undefined) {
|
|
@@ -9,8 +9,8 @@ import type { Api, ApiKey, AssistantMessage, Context, Model, SimpleStreamOptions
|
|
|
9
9
|
import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
|
|
10
10
|
import { prompt } from "@oh-my-pi/pi-utils";
|
|
11
11
|
import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
|
|
12
|
+
import { Tokenizer } from "../tokenizer";
|
|
12
13
|
import type { AgentMessage } from "../types";
|
|
13
|
-
import { estimateTokens } from "./compaction";
|
|
14
14
|
import type { ReadonlySessionManager, SessionEntry } from "./entries";
|
|
15
15
|
import {
|
|
16
16
|
type ConvertToLlm,
|
|
@@ -191,7 +191,9 @@ function getMessageFromEntry(entry: SessionEntry): AgentMessage | undefined {
|
|
|
191
191
|
return createBranchSummaryMessage(entry.summary, entry.fromId, entry.timestamp);
|
|
192
192
|
|
|
193
193
|
case "compaction":
|
|
194
|
-
return createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp,
|
|
194
|
+
return createCompactionSummaryMessage(entry.summary, entry.tokensBefore, entry.timestamp, {
|
|
195
|
+
shortSummary: entry.shortSummary,
|
|
196
|
+
});
|
|
195
197
|
|
|
196
198
|
// These don't contribute to conversation content
|
|
197
199
|
case "thinking_level_change":
|
|
@@ -206,14 +208,14 @@ function getMessageFromEntry(entry: SessionEntry): AgentMessage | undefined {
|
|
|
206
208
|
}
|
|
207
209
|
}
|
|
208
210
|
|
|
209
|
-
function estimateBranchSummaryTokens(message: AgentMessage): number {
|
|
210
|
-
if (message.role !== "toolResult") return
|
|
211
|
+
function estimateBranchSummaryTokens(message: AgentMessage, tokenizer: Tokenizer): number {
|
|
212
|
+
if (message.role !== "toolResult") return tokenizer.countMessage(message);
|
|
211
213
|
const text = message.content
|
|
212
214
|
.filter((c): c is { type: "text"; text: string } => c.type === "text")
|
|
213
215
|
.map(c => c.text)
|
|
214
216
|
.join("");
|
|
215
217
|
if (!text) return 0;
|
|
216
|
-
return
|
|
218
|
+
return tokenizer.countMessage({
|
|
217
219
|
...message,
|
|
218
220
|
content: [{ type: "text", text: truncateToolResultForSummary(text) }],
|
|
219
221
|
});
|
|
@@ -232,7 +234,11 @@ function estimateBranchSummaryTokens(message: AgentMessage): number {
|
|
|
232
234
|
* @param entries - Entries in chronological order
|
|
233
235
|
* @param tokenBudget - Maximum tokens to include (0 = no limit)
|
|
234
236
|
*/
|
|
235
|
-
export function prepareBranchEntries(
|
|
237
|
+
export function prepareBranchEntries(
|
|
238
|
+
entries: SessionEntry[],
|
|
239
|
+
tokenizer: Tokenizer,
|
|
240
|
+
tokenBudget: number = 0,
|
|
241
|
+
): BranchPreparation {
|
|
236
242
|
const messages: AgentMessage[] = [];
|
|
237
243
|
const fileOps = createFileOps();
|
|
238
244
|
let totalTokens = 0;
|
|
@@ -264,7 +270,7 @@ export function prepareBranchEntries(entries: SessionEntry[], tokenBudget: numbe
|
|
|
264
270
|
// Extract file ops from assistant messages (tool calls)
|
|
265
271
|
extractFileOpsFromMessage(message, fileOps);
|
|
266
272
|
|
|
267
|
-
const tokens = estimateBranchSummaryTokens(message);
|
|
273
|
+
const tokens = estimateBranchSummaryTokens(message, tokenizer);
|
|
268
274
|
|
|
269
275
|
// Check budget before adding
|
|
270
276
|
if (tokenBudget > 0 && totalTokens + tokens > tokenBudget) {
|
|
@@ -309,8 +315,9 @@ export async function generateBranchSummary(
|
|
|
309
315
|
// Token budget = context window minus reserved space for prompt + response
|
|
310
316
|
const contextWindow = model.contextWindow || 128000;
|
|
311
317
|
const tokenBudget = contextWindow - reserveTokens;
|
|
318
|
+
const tokenizer = new Tokenizer(model);
|
|
312
319
|
|
|
313
|
-
const { messages, fileOps } = prepareBranchEntries(entries, tokenBudget);
|
|
320
|
+
const { messages, fileOps } = prepareBranchEntries(entries, tokenizer, tokenBudget);
|
|
314
321
|
|
|
315
322
|
if (messages.length === 0) {
|
|
316
323
|
return { summary: "No content to summarize" };
|
|
@@ -25,6 +25,7 @@ import {
|
|
|
25
25
|
} from "@oh-my-pi/pi-ai/providers/openai-shared";
|
|
26
26
|
import { captureOpenAIHttpError } from "@oh-my-pi/pi-ai/utils/openai-http";
|
|
27
27
|
import {
|
|
28
|
+
applyCodexResidencyHeader,
|
|
28
29
|
CODEX_BASE_URL,
|
|
29
30
|
getCodexAccountId,
|
|
30
31
|
OPENAI_HEADER_VALUES,
|
|
@@ -416,6 +417,7 @@ function buildCompactionV2Headers(
|
|
|
416
417
|
if (accountId) {
|
|
417
418
|
headers[OPENAI_HEADERS.ACCOUNT_ID] = accountId;
|
|
418
419
|
}
|
|
420
|
+
applyCodexResidencyHeader(headers, apiKey);
|
|
419
421
|
if (routingSessionId) {
|
|
420
422
|
headers[OPENAI_HEADERS.CONVERSATION_ID] = routingSessionId;
|
|
421
423
|
headers[OPENAI_HEADERS.SESSION_ID] = routingSessionId;
|