pi-codex-compaction 0.1.1 → 0.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -6,6 +6,25 @@ This project follows the spirit of [Keep a Changelog](https://keepachangelog.com
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.1.3] - 2026-09-11
10
+
11
+ ### Fixed
12
+
13
+ - Estimate compaction tokens separately from request bytes so long Codex sessions do not fall back merely because UTF-8 bytes exceed the token budget.
14
+ - Respect request-transform field removals when sizing, restrict warnings to TUI mode, and distinguish preparation failures from transport failures.
15
+ - Preserve the independent 16 MiB request ceiling, count the complete transformed envelope, and reject unavailable context limits.
16
+ - Record safe fallback reasons and numeric size diagnostics in local custom session entries; notify when UI is available without exposing request or provider content.
17
+
18
+ ## [0.1.2] - 2026-09-11
19
+
20
+ ### Fixed
21
+
22
+ - Preserve pi-codex-tools grammar metadata and custom-tool history during remote compaction.
23
+ - Honor pi-fast on direct compaction requests through a public event-bus contract.
24
+ - Include cache writes and standard Pi compaction usage in session totals.
25
+ - Restrict endpoint paths and ports; merge beta features and honor null auth headers.
26
+ - Avoid I/O after pre-cancellation, bound total request time, and close completed SSE streams promptly.
27
+
9
28
  ## [0.1.1] - 2026-08-06
10
29
 
11
30
  ### Fixed
package/CONTRIBUTING.md CHANGED
@@ -29,6 +29,33 @@ Before opening a pull request:
29
29
  - Preserve current-model compaction, context bounds, cancellation, HTTPS, and standard Pi fallback behavior.
30
30
  - Add a regression test when changing wire parsing, request construction, checkpoint rehydration, or model-switch fallback behavior.
31
31
 
32
+ ## Token-budget regression and smoke checks
33
+
34
+ `tests/integration.test.mjs` uses the real Pi compaction hook, serializer, and
35
+ session tree with fake credentials and mocked Responses transport. A synthetic
36
+ long transcript must produce a remote checkpoint even when its UTF-8 bytes
37
+ exceed the numeric token budget. Oversized input must still invoke the standard
38
+ compactor and record a safe fallback reason. A later model request must not
39
+ contain the diagnostic entry. No private session fixtures or live requests are
40
+ needed for these tests.
41
+
42
+ The token conversion follows Codex's ordinary JSON-item heuristic:
43
+ [byte-to-token estimate](https://github.com/openai/codex/blob/654b0a77d0d2f81aa21f61caf7af4be88fe550bb/codex-rs/utils/string/src/truncate.rs)
44
+ and [history sizing](https://github.com/openai/codex/blob/654b0a77d0d2f81aa21f61caf7af4be88fe550bb/codex-rs/core/src/context_manager/history.rs).
45
+ This package keeps all serialized opaque/image bytes in its estimate instead of
46
+ copying Codex's modality-specific discounts. Never equate bytes and tokens or
47
+ remove the independent wire-size ceiling.
48
+
49
+ For an offline replay, reconstruct a compaction's discarded history in memory,
50
+ use fake authentication, replace `fetch` with a fixture checkpoint response,
51
+ and call `createRemoteCompaction`. Print only sizes and the success/fallback
52
+ code. Do not save the transcript, request, auth headers, or opaque content.
53
+
54
+ For an approved live smoke test, follow the procedure in [README.md](./README.md#reference-and-smoke-test).
55
+ Check for `fromHook: true` and `details.kind: "pi-codex-compaction"` on success.
56
+ On fallback, inspect only the reason and counters in the custom diagnostic
57
+ entry. Never inspect or print `encryptedContent`.
58
+
32
59
  ## Code of conduct
33
60
 
34
61
  This project follows the Contributor Covenant Code of Conduct.
package/README.md CHANGED
@@ -9,11 +9,11 @@ Keep long Pi sessions usable on OpenAI Codex models by replacing Pi's local summ
9
9
  - Retains the normal Codex Responses request envelope, including system instructions, active tool schemas, reasoning settings, prompt-cache fields, and routing fields.
10
10
  - Persists Codex's opaque encrypted checkpoint and rehydrates it only for supported Codex requests.
11
11
  - Reuses checkpoints only for the same model, trusted endpoint, Codex account, and authentication mode.
12
- - Bounds input with a UTF-8-aware token estimate, trims tool output when necessary, retries transient failures, and honors cancellation.
12
+ - Bounds input with a Codex-style UTF-8 token estimate and a separate hard byte limit, trims tool output when necessary, retries transient failures, and honors cancellation.
13
13
  - Falls back to standard Pi compaction on failure or when custom compaction instructions are requested.
14
14
  - Keeps a bounded readable transcript excerpt so switching models or providers remains usable.
15
15
 
16
- The current Codex catalog includes models such as `gpt-5.6-sol`, `gpt-5.6-terra`, and `gpt-5.6-luna`. Capability detection follows the provider/API contract (`openai-codex` + `openai-codex-responses`) rather than a brittle model-name list.
16
+ The current Codex catalog includes `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, and `gpt-5.6-luna`. Capability detection follows the provider/API contract (`openai-codex` + `openai-codex-responses`) rather than a brittle model-name list. Use Pi 0.85.1 or later.
17
17
 
18
18
  ## Installation
19
19
 
@@ -39,10 +39,84 @@ When Pi starts compaction on a supported Codex model, the extension sends a stre
39
39
 
40
40
  Compaction uses the model active when Pi triggers it. If a session switches from a larger to a smaller model, the remote request is bounded against the new model's context window and tool outputs are reduced before sending. A previous opaque checkpoint is treated as incompatible after a model, endpoint, account, or authentication-mode switch; Pi's readable previous summary is sent instead. If the full request still cannot fit, the extension leaves compaction to Pi's normal implementation.
41
41
 
42
- If the remote request fails, is cancelled, or returns an unexpected response, Pi's standard compaction path runs. Custom compaction instructions also use Pi's standard path because RemoteCompactionV2 has no documented custom-instructions field. The direct checkpoint request is restricted to `https://chatgpt.com`, rejects redirects, limits request/response size, and never decodes or logs `encrypted_content`. No configuration is required.
42
+ If the remote request fails or returns an unexpected response, Pi's standard compaction path runs. Cancellation remains cancelled. Custom compaction instructions also use Pi's standard path because RemoteCompactionV2 has no documented custom-instructions field. The direct checkpoint request is restricted to `https://chatgpt.com`, rejects redirects, limits request/response size, and never decodes or logs `encrypted_content`. No configuration is required.
43
+
44
+ ### Size limits
45
+
46
+ Input is estimated as `ceil(UTF-8 request bytes / 4)`, with 8,192 tokens reserved
47
+ from the active model's context window. This uses Codex's ordinary-item heuristic,
48
+ not an exact tokenizer. The complete transformed request is counted, including
49
+ system instructions, tool definitions, and routing fields. Opaque checkpoints
50
+ and image data remain counted at their serialized size; they are not decoded or
51
+ discounted. Non-ASCII text uses UTF-8 bytes, not JavaScript string length.
52
+
53
+ The uncompressed request also has an independent **16 MiB hard limit**. Tool
54
+ outputs are reduced only when one of these limits is exceeded. User messages,
55
+ tool calls, and opaque checkpoints are not removed. If the remaining request
56
+ still cannot fit, or the model's context limit is unknown, standard Pi compaction
57
+ runs. The estimate can differ from the server's token count; a server rejection
58
+ still uses the existing fallback.
59
+
60
+ ### Fallback diagnostics
61
+
62
+ A fallback on a supported model records a local custom session entry with type
63
+ `pi-codex-compaction:fallback:v1`. It contains `version: 1` and a reason:
64
+ `custom-instructions`, `auth-unavailable`, `request-unavailable`,
65
+ `context-window-unavailable`, `context-limit`, `request-size-limit`, or
66
+ `remote-failed`. Size failures also include estimated tokens, token budget,
67
+ request bytes, byte limit, and the number of tool outputs reduced.
68
+ Unexpected preparation failures use `request-unavailable`; `remote-failed`
69
+ is reserved for failures from the transport call.
70
+
71
+ No prompt, tool content, encrypted checkpoint, account identifier, credential, or
72
+ raw provider error is included. These entries are not sent to the model. Pi
73
+ shows a warning only in TUI mode when notifications are available; print, JSON,
74
+ and RPC modes get no extra notifications or console output. Unsupported models
75
+ and cancelled attempts do not create fallback diagnostics. Diagnostic storage
76
+ or notification failure does not stop the standard compactor.
43
77
 
44
78
  ## Development
45
79
 
80
+ ### Other Codex extensions
81
+
82
+ With `pi-fast` installed, direct compaction requests use the current Fast toggle.
83
+ With `pi-codex-tools` installed, `apply_patch` keeps its raw grammar definition,
84
+ custom-tool calls, and custom-tool results during compaction. Neither package is
85
+ required. No package reads a private Pi tool registry.
86
+ With `pi-openai-reasoning` installed, verified Astra requests keep the original
87
+ request effort and receive the current effort as a configuration update.
88
+ Failed compaction does not change saved reasoning state.
89
+
90
+ Pi 0.85.1 does not expose grammar metadata in `getAllTools()`. Two synchronous,
91
+ versioned `pi.events` contracts let cooperating extensions supply it:
92
+
93
+ - `pi-codex-compaction:tools:v1`: `{ model, tools }`, before provider serialization.
94
+ A tool owner can attach its own `constrainedSampling` metadata.
95
+ - `pi-codex-compaction:request:v1`: `{ ctx, messages, payload }`, after input
96
+ assembly and before size checks. A listener can replace `payload`. This event
97
+ is not the general `before_provider_request` chain and does not carry auth.
98
+ Size checks use the transformed envelope, including field removals.
99
+
100
+ Other extensions' private request changes are not applied automatically.
101
+ Unknown third-party grammar metadata needs cooperation through the tools event.
102
+ The checkpoint entry includes standard Pi `usage`, including cache reads and
103
+ cache writes, as well as the original bounded token counters in `details`.
104
+ Costs follow Pi's catalog estimates. They are not a ChatGPT subscription bill.
105
+
106
+ ### Reference and smoke test
107
+
108
+ Behavior was checked against `openai/codex` commit
109
+ `654b0a77d0d2f81aa21f61caf7af4be88fe550bb` (2026-09-11), notably
110
+ `core/src/compact_remote_v2{,_attempt}.rs`. No Codex code was copied.
111
+ RemoteCompactionV2 is a changing Codex protocol, not the public `/responses/compact`
112
+ API. Async tools and mid-turn steering require upstream Pi support.
113
+
114
+ For a small live test, load this package and select `openai-codex/gpt-6-astra`.
115
+ Send two short messages, run `/compact`, then ask about the first message.
116
+ Repeat with `/fast on` and `pi-codex-tools` loaded. Check that compaction succeeds,
117
+ the continuation retains context, and session usage includes compaction tokens.
118
+ Use only a temporary file if you test `apply_patch`. Do not generate images.
119
+
46
120
  ```bash
47
121
  npm install
48
122
  npm run -w packages/pi-codex-compaction check
package/SECURITY.md CHANGED
@@ -25,4 +25,28 @@ The extension bounds request, compaction input, and response size, validates the
25
25
 
26
26
  The extension never logs prompts, conversation contents, credentials, authorization headers, or raw provider responses. Install/update telemetry is best-effort and sends only package/version/runtime metadata; it can be disabled with `PI_OFFLINE=1`, `PI_TELEMETRY=0`, or Pi's `enableInstallTelemetry: false` setting.
27
27
 
28
+ Direct requests allow only the official `/backend-api/codex/responses` endpoint
29
+ on the default HTTPS port, without query strings or fragments. Total request
30
+ time is limited to five minutes, including retries. Completed streams are closed
31
+ without waiting for a server disconnect. A pre-aborted request does no network I/O.
32
+ Null auth headers remove matching model headers; beta features are merged.
33
+
34
+ Cooperating local extensions can inspect and transform compaction inputs through
35
+ the documented event bus before size checks. These events contain no credentials.
36
+ They have the same trust level as other installed Pi extensions.
37
+
38
+ Context sizing uses `ceil(UTF-8 serialized request bytes / 4)` with an 8,192-token
39
+ reserve from the active model's context window. It is an estimate, not a strict
40
+ tokenizer bound. The complete transformed envelope is counted, including opaque
41
+ content at its serialized size. A separate 16 MiB uncompressed request limit is
42
+ enforced before network I/O. The HTTPS, redirect, response-size, checkpoint
43
+ compatibility, and cancellation checks remain independent of token estimation.
44
+
45
+ Fallback diagnostics store only a fixed reason code and finite non-negative
46
+ size counters in `pi-codex-compaction:fallback:v1` custom session entries.
47
+ These local records never include credentials, account/model identifiers,
48
+ request content, encrypted checkpoints, or raw errors, and do not enter model
49
+ context. No external diagnostic telemetry is added. They use the existing
50
+ session's permissions and retention policy.
51
+
28
52
  See [CONTRIBUTING.md](./CONTRIBUTING.md) for development and validation instructions.
@@ -7,6 +7,8 @@ import { BETA_FEATURE, getCodexAccountFingerprint } from "../src/codex-wire.js";
7
7
  import { reportInstallTelemetry } from "../src/install-telemetry.js";
8
8
  import {
9
9
  applyRemoteCompactionMarker,
10
+ COMPACTION_FALLBACK_ENTRY,
11
+ type CompactionFallback,
10
12
  createRemoteCompaction,
11
13
  findActiveRemoteCompaction,
12
14
  getCodexAuthKind,
@@ -14,24 +16,69 @@ import {
14
16
  supportsRemoteCompaction,
15
17
  } from "../src/remote-compaction.js";
16
18
 
19
+ const FALLBACK_MESSAGES: Record<CompactionFallback["reason"], string> = {
20
+ "custom-instructions": "custom compaction instructions require the standard compactor",
21
+ "auth-unavailable": "Codex authentication is unavailable",
22
+ "request-unavailable": "the Codex request could not be prepared",
23
+ "context-window-unavailable": "the active model's context limit is unavailable",
24
+ "context-limit": "the estimated input exceeds the active model's token budget",
25
+ "request-size-limit": "the request exceeds the 16 MiB byte limit",
26
+ "remote-failed": "the remote request failed",
27
+ };
28
+
17
29
  export default function piCodexCompaction(pi: ExtensionAPI): void {
18
30
  reportInstallTelemetry();
19
31
 
20
32
  const onBeforeCompact: ExtensionHandler<SessionBeforeCompactEvent, { compaction?: NonNullable<Awaited<ReturnType<typeof createRemoteCompaction>>> }> = async (event, ctx) => {
21
33
  if (!supportsRemoteCompaction(ctx.model)) return undefined;
22
34
 
35
+ let reported = false;
36
+ const reportFallback = (diagnostic: CompactionFallback) => {
37
+ if (reported || event.signal.aborted) return;
38
+ reported = true;
39
+ // No text, model/account identifiers, headers, or provider errors belong
40
+ // in diagnostics. Custom entries do not enter the model's context.
41
+ const data: Record<string, unknown> = { version: 1, reason: diagnostic.reason };
42
+ for (const key of ["estimatedTokens", "tokenBudget", "requestBytes", "byteLimit", "trimmedToolOutputs"] as const) {
43
+ const value = diagnostic[key];
44
+ if (typeof value === "number" && Number.isFinite(value) && value >= 0) data[key] = value;
45
+ }
46
+ try {
47
+ pi.appendEntry(COMPACTION_FALLBACK_ENTRY, data);
48
+ } catch {
49
+ // Diagnostics must not prevent the standard compactor from running.
50
+ }
51
+ if (ctx.mode === "tui" && ctx.hasUI) {
52
+ try {
53
+ ctx.ui.notify(`Codex remote compaction skipped: ${FALLBACK_MESSAGES[diagnostic.reason]}. Using standard Pi compaction.`, "warning");
54
+ } catch {
55
+ // Notification failure must not change compaction behavior either.
56
+ }
57
+ }
58
+ };
23
59
  try {
24
60
  const compaction = await createRemoteCompaction(event, ctx, () => {
25
61
  const active = new Set(pi.getActiveTools());
26
- return pi.getAllTools()
27
- .filter((tool) => active.has(tool.name))
28
- .map(({ name, description, parameters }) => ({ name, description, parameters }));
29
- }, pi.getThinkingLevel?.());
62
+ const data = {
63
+ model: ctx.model,
64
+ tools: pi.getAllTools().filter((tool) => active.has(tool.name)),
65
+ };
66
+ // Pi's public ToolInfo omits constrainedSampling. Tool owners can supply
67
+ // it here without exposing executors or inspecting private registries.
68
+ pi.events?.emit("pi-codex-compaction:tools:v1", data);
69
+ return data.tools;
70
+ }, pi.getThinkingLevel?.(), (payload, messages) => {
71
+ const data = { payload, messages, ctx };
72
+ // Synchronous bus contract: apply pure request transforms before bounds
73
+ // and before network I/O. Never include auth in this event.
74
+ pi.events?.emit("pi-codex-compaction:request:v1", data);
75
+ return data.payload;
76
+ }, reportFallback);
30
77
  return compaction ? { compaction } : undefined;
31
78
  } catch {
32
- if (!event.signal.aborted && ctx.hasUI) {
33
- ctx.ui.notify("Codex remote compaction failed; using standard Pi compaction.", "warning");
34
- }
79
+ // Transport failures already have a reason; other exceptions mean the
80
+ // compaction request could not be prepared or processed.
81
+ reportFallback({ reason: "request-unavailable" });
35
82
  return undefined;
36
83
  }
37
84
  };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-codex-compaction",
3
- "version": "0.1.1",
3
+ "version": "0.1.3",
4
4
  "description": "Use OpenAI Codex RemoteCompactionV2 for supported Pi sessions.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -52,10 +52,10 @@
52
52
  "@earendil-works/pi-coding-agent": "*"
53
53
  },
54
54
  "devDependencies": {
55
- "@earendil-works/pi-ai": "^0.82.1",
56
- "@earendil-works/pi-coding-agent": "^0.82.1",
57
- "@types/node": "^26.1.2",
58
- "tsx": "^4.23.1",
55
+ "@earendil-works/pi-ai": "^0.85.1",
56
+ "@earendil-works/pi-coding-agent": "^0.85.1",
57
+ "@types/node": "^26.2.0",
58
+ "tsx": "^4.23.5",
59
59
  "typescript": "^7.0.2"
60
60
  },
61
61
  "publishConfig": {
package/src/codex-wire.ts CHANGED
@@ -1,10 +1,11 @@
1
1
  import { createHash } from "node:crypto";
2
2
 
3
- const MAX_REQUEST_BYTES = 16 * 1024 * 1024;
3
+ export const MAX_COMPACTION_REQUEST_BYTES = 16 * 1024 * 1024;
4
4
  const MAX_RESPONSE_BYTES = 4 * 1024 * 1024;
5
5
  const MAX_ENCRYPTED_CONTENT_CHARS = 2_000_000;
6
6
  const REQUEST_HEADER_TIMEOUT_MS = 30_000;
7
7
  const REQUEST_IDLE_TIMEOUT_MS = 120_000;
8
+ const REQUEST_TOTAL_TIMEOUT_MS = 300_000;
8
9
  const MAX_RETRIES = 2;
9
10
  const RETRY_BASE_DELAY_MS = 200;
10
11
  const MAX_RETRY_DELAY_MS = 30_000;
@@ -16,10 +17,10 @@ export interface CodexCompactionRequest {
16
17
  model: {
17
18
  id: string;
18
19
  baseUrl: string;
19
- headers?: Record<string, string>;
20
+ headers?: Record<string, string | null>;
20
21
  };
21
22
  apiKey: string;
22
- authHeaders?: Record<string, string>;
23
+ authHeaders?: Record<string, string | null>;
23
24
  sessionId?: string;
24
25
  body: Record<string, unknown>;
25
26
  signal?: AbortSignal;
@@ -30,12 +31,14 @@ export interface CodexCompactionUsage {
30
31
  outputTokens?: number;
31
32
  totalTokens?: number;
32
33
  cachedInputTokens?: number;
34
+ cacheWriteInputTokens?: number;
33
35
  reasoningTokens?: number;
34
36
  }
35
37
 
36
38
  export interface CodexCompactionResult {
37
39
  encryptedContent: string;
38
40
  usage?: CodexCompactionUsage;
41
+ serviceTier?: "default" | "priority" | "flex";
39
42
  }
40
43
 
41
44
  /**
@@ -49,17 +52,19 @@ export async function requestRemoteCompaction(request: CodexCompactionRequest):
49
52
  export async function requestRemoteCompactionWithUsage(
50
53
  request: CodexCompactionRequest,
51
54
  ): Promise<CodexCompactionResult> {
55
+ request.signal?.throwIfAborted();
52
56
  const endpoint = resolveCodexResponsesUrl(request.model.baseUrl);
53
57
  const accountId = extractAccountId(request.apiKey);
54
58
  const headers = buildHeaders(request, accountId);
55
59
  const body = JSON.stringify(request.body);
56
- if (new TextEncoder().encode(body).byteLength > MAX_REQUEST_BYTES) {
60
+ if (Buffer.byteLength(body, "utf8") > MAX_COMPACTION_REQUEST_BYTES) {
57
61
  throw new Error("Codex compaction request exceeded the size limit");
58
62
  }
59
63
 
60
64
  const requestState = requestSignal(request.signal);
61
65
  try {
62
66
  for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
67
+ requestState.signal.throwIfAborted();
63
68
  requestState.startAttempt();
64
69
 
65
70
  try {
@@ -116,8 +121,14 @@ export function resolveCodexResponsesUrl(baseUrl: string): string {
116
121
  if (url.username || url.password) {
117
122
  throw new Error("Codex compaction endpoint must not contain URL credentials");
118
123
  }
124
+ if (url.port || url.search || url.hash) {
125
+ throw new Error("Codex compaction endpoint must use the default port without query or fragment");
126
+ }
119
127
 
120
128
  const path = url.pathname.replace(/\/+$/, "");
129
+ if (!["/backend-api", "/backend-api/codex", "/backend-api/codex/responses"].includes(path)) {
130
+ throw new Error("Codex compaction endpoint must use the Codex Responses path");
131
+ }
121
132
  if (path.endsWith("/codex/responses")) {
122
133
  return url.toString();
123
134
  }
@@ -140,14 +151,20 @@ export function parseCompactionSseResult(sse: string): CodexCompactionResult {
140
151
 
141
152
  function buildHeaders(request: CodexCompactionRequest, accountId: string): Headers {
142
153
  const headers = new Headers();
143
- for (const [name, value] of Object.entries(request.model.headers ?? {})) headers.set(name, value);
144
- for (const [name, value] of Object.entries(request.authHeaders ?? {})) headers.set(name, value);
154
+ for (const [name, value] of Object.entries(request.model.headers ?? {})) {
155
+ if (value !== null) headers.set(name, value);
156
+ }
157
+ for (const [name, value] of Object.entries(request.authHeaders ?? {})) {
158
+ if (value !== null) headers.set(name, value);
159
+ else headers.delete(name);
160
+ }
145
161
 
146
162
  headers.set("Authorization", `Bearer ${request.apiKey}`);
147
163
  headers.set("chatgpt-account-id", accountId);
148
164
  headers.set("originator", "pi");
149
165
  headers.set("OpenAI-Beta", "responses=experimental");
150
- headers.set("x-codex-beta-features", BETA_FEATURE);
166
+ const features = headers.get("x-codex-beta-features")?.split(",").map((part) => part.trim()).filter(Boolean) ?? [];
167
+ headers.set("x-codex-beta-features", [...new Set([...features, BETA_FEATURE])].join(","));
151
168
  headers.set("accept", "text/event-stream");
152
169
  headers.set("content-type", "application/json");
153
170
 
@@ -208,6 +225,7 @@ async function readCompactionSse(
208
225
  throw new CodexCompactionError("Codex compaction response exceeded the size limit", false);
209
226
  }
210
227
  parser.push(decoder.decode(value, { stream: true }));
228
+ if (parser.isComplete) return parser.finish();
211
229
  }
212
230
  parser.push(decoder.decode());
213
231
  return parser.finish();
@@ -227,6 +245,11 @@ class CompactionSseParser {
227
245
  private status: string | undefined;
228
246
  private compactionItems: Array<string | undefined> = [];
229
247
  private usage: CodexCompactionUsage | undefined;
248
+ private serviceTier: CodexCompactionResult["serviceTier"];
249
+
250
+ get isComplete(): boolean {
251
+ return this.completed;
252
+ }
230
253
 
231
254
  push(chunk: string): void {
232
255
  this.buffer += chunk;
@@ -262,6 +285,7 @@ class CompactionSseParser {
262
285
  return {
263
286
  encryptedContent,
264
287
  ...(this.usage ? { usage: this.usage } : {}),
288
+ ...(this.serviceTier ? { serviceTier: this.serviceTier } : {}),
265
289
  };
266
290
  }
267
291
 
@@ -294,6 +318,8 @@ class CompactionSseParser {
294
318
  const response = isRecord(event.response) ? event.response : undefined;
295
319
  this.status = typeof response?.status === "string" ? response.status : undefined;
296
320
  this.usage = parseUsage(response?.usage ?? event.usage);
321
+ const tier = response?.service_tier;
322
+ this.serviceTier = tier === "default" || tier === "priority" || tier === "flex" ? tier : undefined;
297
323
 
298
324
  // Codex emits output_item.done before response.completed. Retain this
299
325
  // fallback for compact Responses implementations that only return the
@@ -331,12 +357,14 @@ function parseUsage(value: unknown): CodexCompactionUsage | undefined {
331
357
  const inputDetails = isRecord(value.input_tokens_details) ? value.input_tokens_details : undefined;
332
358
  const outputDetails = isRecord(value.output_tokens_details) ? value.output_tokens_details : undefined;
333
359
  const cachedInputTokens = finiteNonNegativeNumber(inputDetails?.cached_tokens);
360
+ const cacheWriteInputTokens = finiteNonNegativeNumber(inputDetails?.cache_write_tokens);
334
361
  const reasoningTokens = finiteNonNegativeNumber(outputDetails?.reasoning_tokens);
335
362
  if (
336
363
  inputTokens === undefined &&
337
364
  outputTokens === undefined &&
338
365
  totalTokens === undefined &&
339
366
  cachedInputTokens === undefined &&
367
+ cacheWriteInputTokens === undefined &&
340
368
  reasoningTokens === undefined
341
369
  ) {
342
370
  return undefined;
@@ -346,6 +374,7 @@ function parseUsage(value: unknown): CodexCompactionUsage | undefined {
346
374
  ...(outputTokens !== undefined ? { outputTokens } : {}),
347
375
  ...(totalTokens !== undefined ? { totalTokens } : {}),
348
376
  ...(cachedInputTokens !== undefined ? { cachedInputTokens } : {}),
377
+ ...(cacheWriteInputTokens !== undefined ? { cacheWriteInputTokens } : {}),
349
378
  ...(reasoningTokens !== undefined ? { reasoningTokens } : {}),
350
379
  };
351
380
  }
@@ -428,6 +457,10 @@ interface RequestSignal {
428
457
 
429
458
  function requestSignal(signal: AbortSignal | undefined): RequestSignal {
430
459
  const controller = new AbortController();
460
+ const totalTimeout = setTimeout(
461
+ () => controller.abort(new Error("Codex compaction request exceeded the total time limit")),
462
+ REQUEST_TOTAL_TIMEOUT_MS,
463
+ );
431
464
  let headerTimeout: ReturnType<typeof setTimeout> | undefined;
432
465
  let idleTimeout: ReturnType<typeof setTimeout> | undefined;
433
466
 
@@ -468,6 +501,7 @@ function requestSignal(signal: AbortSignal | undefined): RequestSignal {
468
501
  },
469
502
  touch,
470
503
  cleanup: () => {
504
+ clearTimeout(totalTimeout);
471
505
  clearTimers();
472
506
  signal?.removeEventListener("abort", onAbort);
473
507
  },
@@ -1,10 +1,11 @@
1
1
  import { stream as captureProviderPayload } from "@earendil-works/pi-ai/compat";
2
- import type { Model, Tool } from "@earendil-works/pi-ai";
2
+ import { calculateCost, type Model, type Tool, type Usage } from "@earendil-works/pi-ai";
3
3
  import type { ExtensionContext, SessionBeforeCompactEvent } from "@earendil-works/pi-coding-agent";
4
4
  import { convertToLlm, serializeConversation, sessionEntryToContextMessages } from "@earendil-works/pi-coding-agent";
5
5
  import type { SessionEntry } from "@earendil-works/pi-coding-agent";
6
6
  import {
7
7
  getCodexAccountFingerprint,
8
+ MAX_COMPACTION_REQUEST_BYTES,
8
9
  requestRemoteCompactionWithUsage,
9
10
  resolveCodexResponsesUrl,
10
11
  type CodexCompactionUsage,
@@ -12,10 +13,24 @@ import {
12
13
 
13
14
  export const REMOTE_SUMMARY_MARKER = "[pi-codex-compaction:v1]";
14
15
  export const REMOTE_COMPACTION_KIND = "pi-codex-compaction";
16
+ export const COMPACTION_FALLBACK_ENTRY = "pi-codex-compaction:fallback:v1";
17
+
18
+ export interface CompactionFallback {
19
+ reason: "custom-instructions" | "auth-unavailable" | "request-unavailable" |
20
+ "context-window-unavailable" | "context-limit" | "request-size-limit" | "remote-failed";
21
+ estimatedTokens?: number;
22
+ tokenBudget?: number;
23
+ requestBytes?: number;
24
+ byteLimit?: number;
25
+ trimmedToolOutputs?: number;
26
+ }
15
27
 
16
28
  const FALLBACK_SUMMARY_MAX_CHARS = 12_000;
17
29
  const MAX_ENCRYPTED_CONTENT_CHARS = 2_000_000;
18
30
  const COMPACTION_RESPONSE_RESERVE_TOKENS = 8_192;
31
+ // Codex's ordinary-item estimate uses ceil(UTF-8 bytes / 4), not bytes as
32
+ // tokens. This is a heuristic, not a tokenizer or a context-fit guarantee.
33
+ const APPROX_BYTES_PER_TOKEN = 4;
19
34
  const TRUNCATED_TOOL_OUTPUT = "[Tool output omitted from the Codex compaction request to fit the active model context window.]";
20
35
  const PI_COMPACTION_SUMMARY_PREFIX = "The conversation history before this point was compacted into the following summary:";
21
36
 
@@ -95,23 +110,31 @@ export async function createRemoteCompaction(
95
110
  ctx: ExtensionContext,
96
111
  getTools: () => readonly ToolInfoLike[],
97
112
  thinkingLevel?: string,
113
+ prepareRequest?: (payload: Record<string, unknown>, messages: readonly AgentMessage[]) => Record<string, unknown>,
114
+ onFallback?: (diagnostic: CompactionFallback) => void,
98
115
  ): Promise<{
99
116
  summary: string;
100
117
  firstKeptEntryId: string;
101
118
  tokensBefore: number;
102
119
  details: RemoteCompactionDetails;
120
+ usage?: Usage;
103
121
  } | undefined> {
122
+ if (event.signal.aborted) return undefined;
123
+ const skip = (reason: CompactionFallback["reason"]) => {
124
+ onFallback?.({ reason });
125
+ return undefined;
126
+ };
104
127
  // Pi's custom focus is part of the standard summarizer contract. The
105
128
  // Responses compaction envelope has no documented equivalent, so do not
106
129
  // silently discard it.
107
- if (event.customInstructions?.trim()) return undefined;
130
+ if (event.customInstructions?.trim()) return skip("custom-instructions");
108
131
 
109
132
  const model = ctx.model as CodexModel | undefined;
110
133
  if (!model || !supportsRemoteCompaction(model)) return undefined;
111
134
 
112
135
  const endpoint = resolveCodexResponsesUrl(model.baseUrl);
113
136
  const auth = await ctx.modelRegistry.getApiKeyAndHeaders(model as Model<any>);
114
- if (!auth.ok || !auth.apiKey) return undefined;
137
+ if (!auth.ok || !auth.apiKey) return skip("auth-unavailable");
115
138
  const accountFingerprint = getCodexAccountFingerprint(auth.apiKey);
116
139
  const authKind = getCodexAuthKind(ctx.modelRegistry, model as Model<any>);
117
140
 
@@ -133,24 +156,29 @@ export async function createRemoteCompaction(
133
156
  event.signal,
134
157
  thinkingLevel,
135
158
  );
136
- if (!providerPayload) return undefined;
159
+ if (!providerPayload) return skip("request-unavailable");
137
160
  const providerInput = providerPayload.input;
138
- if (!Array.isArray(providerInput)) return undefined;
161
+ if (!Array.isArray(providerInput)) return skip("request-unavailable");
139
162
  const providerTools = Array.isArray(providerPayload.tools) ? providerPayload.tools : [];
140
163
 
141
164
  const input = appendCompactionItems(providerInput, event.preparation, compatiblePrevious);
142
- const requestBody = {
165
+ const requestBody = prepareRequest?.({
143
166
  ...providerPayload,
167
+ input,
168
+ }, messages) ?? { ...providerPayload, input };
169
+ Object.assign(requestBody, {
144
170
  model: model.id,
145
171
  store: false,
146
172
  stream: true,
147
- };
173
+ });
174
+ if (!Array.isArray(requestBody.input)) return skip("request-unavailable");
148
175
  const boundedInput = boundCompactionInput(
149
- input,
176
+ requestBody.input,
150
177
  instructions,
151
- providerTools,
178
+ Array.isArray(requestBody.tools) ? requestBody.tools : providerTools,
152
179
  model.contextWindow,
153
180
  requestBody,
181
+ onFallback,
154
182
  );
155
183
  if (!boundedInput) return undefined;
156
184
 
@@ -164,6 +192,11 @@ export async function createRemoteCompaction(
164
192
  input: boundedInput,
165
193
  },
166
194
  signal: event.signal,
195
+ }).catch((error: unknown) => {
196
+ // Only failures from the transport path receive this reason. Preparation
197
+ // failures are classified by the extension's outer fallback handler.
198
+ if (!event.signal.aborted) onFallback?.({ reason: "remote-failed" });
199
+ throw error;
167
200
  });
168
201
 
169
202
  const fallback = buildFallbackSummary(event.preparation, messages);
@@ -171,6 +204,11 @@ export async function createRemoteCompaction(
171
204
  summary: `${REMOTE_SUMMARY_MARKER}\n\n${fallback}`,
172
205
  firstKeptEntryId: event.preparation.firstKeptEntryId,
173
206
  tokensBefore: event.preparation.tokensBefore,
207
+ ...(result.usage ? { usage: toPiUsage(
208
+ result.usage,
209
+ ctx.model!,
210
+ result.serviceTier && result.serviceTier !== "default" ? result.serviceTier : requestBody.service_tier,
211
+ ) } : {}),
174
212
  details: {
175
213
  kind: REMOTE_COMPACTION_KIND,
176
214
  version: 2,
@@ -200,16 +238,25 @@ export function boundCompactionInput(
200
238
  tools: readonly unknown[],
201
239
  contextWindow: number,
202
240
  requestPayload?: Record<string, unknown>,
241
+ onLimit?: (diagnostic: CompactionFallback) => void,
203
242
  ): unknown[] | undefined {
204
- const budget = Math.max(1, Math.floor(contextWindow - COMPACTION_RESPONSE_RESERVE_TOKENS));
205
- // A byte-level bound is conservative when the active model's tokenizer is unavailable:
206
- // a token can be represented by a single UTF-8 byte, but not fewer.
207
- const budgetBytes = budget;
243
+ if (!Number.isFinite(contextWindow) || contextWindow <= 0) {
244
+ onLimit?.({ reason: "context-window-unavailable" });
245
+ return undefined;
246
+ }
247
+ const budgetTokens = Math.max(0, Math.floor(contextWindow) - COMPACTION_RESPONSE_RESERVE_TOKENS);
208
248
  const bounded = input.map((item) => item);
209
- let requestBytes = estimateConservativeBytes(
249
+ let requestBytes = serializedBytes(
210
250
  buildCompactionRequest(requestPayload, bounded, instructions, tools),
211
251
  );
212
- if (requestBytes <= budgetBytes) return bounded;
252
+ if (!Number.isFinite(requestBytes)) {
253
+ onLimit?.({ reason: "request-unavailable" });
254
+ return undefined;
255
+ }
256
+ const fits = () => requestBytes <= MAX_COMPACTION_REQUEST_BYTES &&
257
+ Math.ceil(requestBytes / APPROX_BYTES_PER_TOKEN) <= budgetTokens;
258
+ if (fits()) return bounded;
259
+ let trimmedToolOutputs = 0;
213
260
 
214
261
  // ponytail: trim tool outputs first; if structural content still exceeds the active model window,
215
262
  // let Pi's standard compaction path handle the request instead of inventing a lossy transcript rewrite.
@@ -218,11 +265,24 @@ export function boundCompactionInput(
218
265
  const replacement = trimToolOutput(item);
219
266
  if (!replacement) continue;
220
267
 
221
- requestBytes += estimateConservativeBytes(replacement) - estimateConservativeBytes(item);
268
+ const removedBytes = serializedBytes(item) - serializedBytes(replacement);
269
+ // A short output can be smaller than the omission notice. Never make a
270
+ // request larger while trying to fit either limit.
271
+ if (removedBytes <= 0) continue;
272
+ requestBytes -= removedBytes;
222
273
  bounded[index] = replacement;
223
- if (requestBytes <= budgetBytes) return bounded;
274
+ trimmedToolOutputs++;
275
+ if (fits()) return bounded;
224
276
  }
225
277
 
278
+ onLimit?.({
279
+ reason: requestBytes > MAX_COMPACTION_REQUEST_BYTES ? "request-size-limit" : "context-limit",
280
+ estimatedTokens: Math.ceil(requestBytes / APPROX_BYTES_PER_TOKEN),
281
+ tokenBudget: budgetTokens,
282
+ requestBytes,
283
+ byteLimit: MAX_COMPACTION_REQUEST_BYTES,
284
+ trimmedToolOutputs,
285
+ });
226
286
  return undefined;
227
287
  }
228
288
 
@@ -247,7 +307,7 @@ async function captureCodexPayload(
247
307
  messages: ReturnType<typeof convertToLlm>,
248
308
  tools: readonly ToolInfoLike[],
249
309
  apiKey: string,
250
- authHeaders: Record<string, string> | undefined,
310
+ authHeaders: Record<string, string | null> | undefined,
251
311
  sessionId: string,
252
312
  signal: AbortSignal,
253
313
  thinkingLevel: string | undefined,
@@ -475,10 +535,12 @@ function buildCompactionRequest(
475
535
  instructions: string,
476
536
  tools: readonly unknown[],
477
537
  ): Record<string, unknown> {
478
- const payload: Record<string, unknown> = requestPayload ? { ...requestPayload } : {};
479
- payload.instructions = instructions;
538
+ // A supplied envelope is authoritative, including fields a listener removed.
539
+ // Separate defaults apply only to callers without a complete payload.
540
+ const payload: Record<string, unknown> = requestPayload
541
+ ? { ...requestPayload }
542
+ : { instructions, ...(tools.length > 0 ? { tools } : {}) };
480
543
  payload.input = input;
481
- if (tools.length > 0 || requestPayload && "tools" in requestPayload) payload.tools = tools;
482
544
  return payload;
483
545
  }
484
546
 
@@ -499,10 +561,10 @@ function trimToolOutput(value: unknown): unknown | undefined {
499
561
  return undefined;
500
562
  }
501
563
 
502
- function estimateConservativeBytes(value: unknown): number {
564
+ function serializedBytes(value: unknown): number {
503
565
  try {
504
566
  const json = JSON.stringify(value) ?? "";
505
- return new TextEncoder().encode(json).byteLength;
567
+ return Buffer.byteLength(json, "utf8");
506
568
  } catch {
507
569
  return Number.POSITIVE_INFINITY;
508
570
  }
@@ -534,7 +596,27 @@ function limitText(text: string, maxChars: number): string {
534
596
  return `${text.slice(0, head)}${omission}${text.slice(text.length - (available - head))}`;
535
597
  }
536
598
 
537
- type ToolInfoLike = Pick<Tool, "name" | "description" | "parameters">;
599
+ type ToolInfoLike = Tool;
600
+
601
+ function toPiUsage(raw: CodexCompactionUsage, model: Model<any>, tier: unknown): Usage {
602
+ const cacheRead = raw.cachedInputTokens ?? 0;
603
+ const cacheWrite = raw.cacheWriteInputTokens ?? 0;
604
+ const output = raw.outputTokens ?? 0;
605
+ const usage: Usage = {
606
+ input: Math.max(0, (raw.inputTokens ?? 0) - cacheRead - cacheWrite),
607
+ output,
608
+ cacheRead,
609
+ cacheWrite,
610
+ ...(raw.reasoningTokens !== undefined ? { reasoning: raw.reasoningTokens } : {}),
611
+ totalTokens: raw.totalTokens ?? (raw.inputTokens ?? 0) + output,
612
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
613
+ };
614
+ calculateCost(model, usage);
615
+ // Match Pi 0.85.1's catalog estimates, not subscription credit accounting.
616
+ const multiplier = tier === "priority" ? (model.id === "gpt-5.5" ? 2.5 : 2) : tier === "flex" ? 0.5 : 1;
617
+ for (const key of ["input", "output", "cacheRead", "cacheWrite", "total"] as const) usage.cost[key] *= multiplier;
618
+ return usage;
619
+ }
538
620
 
539
621
  function isRecord(value: unknown): value is Record<string, unknown> {
540
622
  return typeof value === "object" && value !== null && !Array.isArray(value);