pi-openai-codex-compat 0.0.8 → 0.0.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,14 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
+ ## 0.0.9 - 2026-08-16
6
+
7
+ ### Fixed
8
+
9
+ - Preserve deferred incomplete and failed Codex response handling across linked tool execution without requiring session affinity or Pi agent-turn hooks.
10
+ - Preserve provider items committed before a later context-overflow subrequest, compact the validated prefix, and retry automatically from the native checkpoint.
11
+ - Split successful Codex follow-up sampling at percentage-compaction boundaries so Pi records the committed prefix, native checkpoint, and continued response in chronological order without synthetic model input.
12
+
5
13
  ## 0.0.8 - 2026-08-16
6
14
 
7
15
  ### Fixed
package/README.md CHANGED
@@ -35,19 +35,19 @@ The compatibility baseline is official Codex CLI `0.146.0`, released July 29, 20
35
35
 
36
36
  ### Configurable defaults that differ from Codex
37
37
 
38
- | Area | This package by default | Official Codex | Configuration |
39
- | --------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------ |
40
- | Generated-image detail sent back to the model | Sends image tool-result content with `input_image.detail: "auto"`. On GPT-5.6, `auto` uses original-size image accounting. | Uses `high`. | `imageDetail`: `auto`, `low`, `high`, or `original`. |
41
- | Image-generation tool | Enabled whenever an `openai-codex` model is selected. Backend capability and account failures surface when the tool executes. | Stable and enabled by default, but additionally gated by plan, model, provider, authentication, image-generation, and namespace capabilities. | `imageGeneration`: boolean. |
42
- | Standalone `web.run` | Disabled by default; when enabled, preferred over hosted `web_search` and sent with the complete reserved schema and description. | Enabled by default for `gpt-5.6-sol` through Responses Lite; otherwise subject to standalone-search feature and runtime gates. | `webRun`: boolean. |
43
- | Hosted web search | Disabled by default; when enabled, injected only for ordinary Responses while `web.run` is inactive. Responses Lite omits hosted tools. | Omitted for `gpt-5.6-sol` while standalone `web.run` is available; otherwise defaults to cached mode when hosted search is supported. | `webRun` and `webSearch`: `disabled`, `cached`, `indexed`, or `live`. |
44
- | Coding mutation tools | Enables `apply_patch` and suppresses Pi's active `edit` and `write` tools. | Chooses its tool surface from model metadata and runtime capabilities; there are no Pi `edit` or `write` tools to suppress. | `applyPatch`: boolean. |
45
- | `apply_patch` debug output | Disabled; collapsed results show the normal visual summary and instruction rows. | Not applicable to Pi's tool-result renderer. | `applyPatchDebug`: boolean. |
46
- | Codex tool background | Uses a subtle theme-derived surface for extension-owned Codex tools. | Uses Codex's own TUI activity cells rather than Pi tool rows. | `toolBackground`: `subtle`, `status`, or `none`. |
47
- | Auto-compaction trigger | Relies on Pi's reserve-token threshold unless a percentage is configured. | Tracks Codex's model/token-budget state before and between sampling steps. | `autoCompactAtPercent`: percentage or unset. Pi's own compaction settings remain separate. |
48
- | Fast mode | Uses the normal tier. | Uses the configured Codex service tier. | `fastMode`: boolean; `true` requests the priority tier. |
49
- | Responses Lite | Disabled; supported GPT-5.6 models use ordinary Responses. | Enabled according to Codex model metadata. | `responsesLite`: boolean; `true` enables Responses Lite. |
50
- | Text and reasoning request controls | Sends low text verbosity and automatic reasoning summaries; omits the default GPT-5.6 standard mode and sends `reasoning.mode` only for pro mode. | Resolves these controls through Codex configuration, model metadata, and turn state. | `textVerbosity`, `reasoningSummary`, and `reasoningMode`. |
38
+ | Area | This package by default | Official Codex | Configuration |
39
+ | --------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
40
+ | Generated-image detail sent back to the model | Sends image tool-result content with `input_image.detail: "auto"`. On GPT-5.6, `auto` uses original-size image accounting. | Uses `high`. | `imageDetail`: `auto`, `low`, `high`, or `original`. |
41
+ | Image-generation tool | Enabled whenever an `openai-codex` model is selected. Backend capability and account failures surface when the tool executes. | Stable and enabled by default, but additionally gated by plan, model, provider, authentication, image-generation, and namespace capabilities. | `imageGeneration`: boolean. |
42
+ | Standalone `web.run` | Disabled by default; when enabled, preferred over hosted `web_search` and sent with the complete reserved schema and description. | Enabled by default for `gpt-5.6-sol` through Responses Lite; otherwise subject to standalone-search feature and runtime gates. | `webRun`: boolean. |
43
+ | Hosted web search | Disabled by default; when enabled, injected only for ordinary Responses while `web.run` is inactive. Responses Lite omits hosted tools. | Omitted for `gpt-5.6-sol` while standalone `web.run` is available; otherwise defaults to cached mode when hosted search is supported. | `webRun` and `webSearch`: `disabled`, `cached`, `indexed`, or `live`. |
44
+ | Coding mutation tools | Enables `apply_patch` and suppresses Pi's active `edit` and `write` tools. | Chooses its tool surface from model metadata and runtime capabilities; there are no Pi `edit` or `write` tools to suppress. | `applyPatch`: boolean. |
45
+ | `apply_patch` debug output | Disabled; collapsed results show the normal visual summary and instruction rows. | Not applicable to Pi's tool-result renderer. | `applyPatchDebug`: boolean. |
46
+ | Codex tool background | Uses a subtle theme-derived surface for extension-owned Codex tools. | Uses Codex's own TUI activity cells rather than Pi tool rows. | `toolBackground`: `subtle`, `status`, or `none`. |
47
+ | Auto-compaction trigger | Relies on Pi's reserve-token threshold unless a percentage is configured. | Tracks Codex's model/token-budget state before and between sampling steps. | `autoCompactAtPercent`: percentage or unset. Mid-response percentage boundaries use Pi's bounded compact-and-continue lifecycle, so Pi auto-compaction must remain enabled. |
48
+ | Fast mode | Uses the normal tier. | Uses the configured Codex service tier. | `fastMode`: boolean; `true` requests the priority tier. |
49
+ | Responses Lite | Disabled; supported GPT-5.6 models use ordinary Responses. | Enabled according to Codex model metadata. | `responsesLite`: boolean; `true` enables Responses Lite. |
50
+ | Text and reasoning request controls | Sends low text verbosity and automatic reasoning summaries; omits the default GPT-5.6 standard mode and sends `reasoning.mode` only for pro mode. | Resolves these controls through Codex configuration, model metadata, and turn state. | `textVerbosity`, `reasoningSummary`, and `reasoningMode`. |
51
51
 
52
52
  `web.run` is a reserved GPT-5.6 tool name. Its declaration therefore reproduces the complete current Codex post-normalization `SearchCommands` schema and official tool description instead of using Pi's normal compact tool schema. This intentionally omits generated annotations such as `format` and `minimum` that Codex removes before sending the declaration to Responses.
53
53
 
@@ -61,9 +61,9 @@ The compatibility baseline is official Codex CLI `0.146.0`, released July 29, 20
61
61
  | System instructions | Pi rebuilds the current system prompt. Responses Lite models prepend it as developer input after `additional_tools`; other models send it through Responses `instructions`. Normal Pi history does not store it as replayed system/developer input. `/reload` updates the next request without rewriting old checkpoints. |
62
62
  | Turn metadata | Requests send a persisted installation id plus Pi-derived session, thread, context-window, turn, source, sandbox, request-kind, and nested compaction-operation metadata in `client_metadata` and compatible headers. The in-memory context-window number advances after successful compaction. One turn id is reused throughout a Pi agent run, while prewarm has its own id. First-party requests also carry Codex's model-and-tier routing hint. The provider captures the server-issued `x-codex-turn-state` once per agent run, replays it on WebSocket retries, SSE requests, and WebSocket-to-SSE fallback, and records all identity values in transport diagnostics. Pi does not reconstruct prior window number after extension reload/session resume or reproduce workspace Git/parent/subagent/Code Mode metadata. Each marked Pi tree branch receives its own persisted thread UUID. |
63
63
  | Cache preparation | Before the first cache-enabled WebSocket turn, the package prewarms only the stable instruction/tool prefix: ordinary Responses uses empty `input`, while Responses Lite uses `additional_tools` plus the developer instructions. The first generated request then contributes only dynamic conversation input to the continuation. No explicit prompt-cache breakpoints are added. |
64
- | Mid-turn compaction | Provider-boundary percentage compaction installs a checkpoint and continues the intercepted request. Pi threshold compaction normally runs after the agent response; after Codex output-token truncation, the extension queues a hidden continuation so threshold compaction completes before sampling resumes. Official Codex owns this sampling and compaction loop directly. |
64
+ | Mid-turn compaction | Provider-boundary percentage compaction preserves a successful `end_turn:false` prefix as its own Pi assistant message, installs a checkpoint, and continues without synthetic model input. Pi threshold compaction normally runs after the agent response; after Codex output-token truncation, the extension queues a hidden continuation so threshold compaction completes before sampling resumes. Official Codex owns this sampling and compaction loop directly. |
65
65
  | Provider-owned follow-up | Completed responses with `end_turn: false` continue immediately from completed native output without synthetic user input. Retryable `response.failed` and all `response.incomplete` events are resampled with the official five-retry stream budget, preserving completed output and cumulative usage while excluding unfinished attempt content. A `max_output_tokens` response that exhausts this budget still becomes Pi `stopReason: "length"` and uses the extension's unbounded host-level continuation recovery. |
66
- | Compaction lifecycle events | Percentage compaction writes through Pi's mutable session manager but cannot emit Pi's internal `session_compact` event through the public extension API. Manual, threshold, and overflow compactions initiated by Pi do emit the normal lifecycle. |
66
+ | Compaction lifecycle events | Pre-turn percentage compaction writes through Pi's mutable session manager and cannot emit Pi's internal `session_compact` event through the public extension API. Mid-response percentage boundaries and manual, threshold, or overflow compactions initiated by Pi emit the normal lifecycle. |
67
67
  | Header hooks | An internal percentage-compaction request reuses the already transformed provider headers. It cannot independently rerun Pi's `before_provider_headers` hook. |
68
68
  | Native retained context | Deliberately differs from current Codex. The package retains recent user/developer/system messages under the 64k budget before the opaque compaction item. Current Codex applies a second installed-history filter that drops developer/system wrappers and non-real-user messages, can retain eligible structured agent commentary, and trims oversized function outputs before compaction. Pi keeps its existing checkpoint shape by design. |
69
69
  | Tool namespaces | Responses Lite groups Pi's ordinary function/custom declarations into upstream's canonical `functions` namespace and maps that default namespace back to bare Pi names. Pi registers dotted names such as `web.run` as exact flat identifiers, so the provider converts only the fixed extension-owned allowlist into non-default Responses namespace/member identities and rejects unknown or ambiguously flat namespaced calls. |
@@ -198,7 +198,7 @@ Defaults:
198
198
  | `imageGeneration` | boolean | `true` | Enables the extension-owned `image_gen.imagegen` tool on selected `openai-codex` models. |
199
199
  | `imageDetail` | `auto`, `low`, `high`, `original` | `auto` | Sets `input_image.detail` when an image tool result is sent back to the model. It does not change `gpt-image-2` generation quality. |
200
200
  | `webRun` | boolean | `false` | Enables the extension-owned `web.run` tool on selected `openai-codex` models. When active, it replaces hosted `web_search` in the Responses tool list. |
201
- | `autoCompactAtPercent` | number greater than `0` and at most `100`, or `null` | unset | Adds provider-boundary compaction independently of Pi's normal reserve-token threshold. A project value of `null` disables a global percentage threshold. |
201
+ | `autoCompactAtPercent` | number greater than `0` and at most `100`, or `null` | unset | Adds provider-boundary compaction independently of Pi's normal reserve-token threshold. Mid-response boundaries require Pi auto-compaction. A project value of `null` disables a global percentage threshold. |
202
202
  | `webSearch` | `disabled`, `cached`, `indexed`, `live` | `disabled` | Controls hosted search and standalone-search external access. `disabled` removes hosted search but leaves an independently enabled `web.run` in cached-only mode; `indexed` prefers indexed content; `live` permits live external access. |
203
203
  | `textVerbosity` | `low`, `medium`, `high` | `low` | Sets Responses API `text.verbosity`. |
204
204
  | `reasoningSummary` | `auto`, `concise`, `detailed`, `off` | `auto` | Sets `reasoning.summary` when reasoning is enabled; `off` omits the summary parameter. |
@@ -61,7 +61,11 @@ import {
61
61
  type CodexWebSocketResponseHandle,
62
62
  } from "./codex-transport.ts";
63
63
  import { DEFAULT_CONFIG, type CodexCompatConfig, type ImageDetail } from "./config.ts";
64
- import { nativeResponseData, NATIVE_RESPONSE_ENTRY_TYPE } from "./native-history.ts";
64
+ import {
65
+ nativeResponseData,
66
+ NATIVE_RESPONSE_ENTRY_TYPE,
67
+ type NativeResponseAttempt,
68
+ } from "./native-history.ts";
65
69
  import {
66
70
  CODEX_NAMESPACED_TOOL_NAMES,
67
71
  CODEX_TEXT_CONTENT_ITEM_TOOL_RESULT_NAMES,
@@ -78,6 +82,7 @@ import { formatProviderError } from "./provider-error.ts";
78
82
  const CODEX_PROVIDER = "openai-codex";
79
83
  const CODEX_API = "openai-codex-responses";
80
84
  const CODEX_TOOL_CALL_PROVIDERS = new Set(["openai", "openai-codex", "opencode"]);
85
+ const PROVIDER_COMPACTION_BOUNDARY_STOP_REASON = "completed.end_turn_false.context_limit";
81
86
 
82
87
  type ConfigResolver = (ctx: ExtensionContext) => CodexCompatConfig;
83
88
 
@@ -115,7 +120,6 @@ type ActiveAgentTurn = {
115
120
  turnId: string;
116
121
  startedAtUnixMs: number;
117
122
  turnState: CodexTurnState;
118
- pendingPostToolDisposition?: CodexPostToolDisposition;
119
123
  };
120
124
 
121
125
  type CodexCompat = {
@@ -145,13 +149,16 @@ type CodexPostToolDisposition = {
145
149
  callIds: string[];
146
150
  response?: JsonRecord;
147
151
  retryAttempt: number;
152
+ sessionId?: string;
148
153
  terminalType: "response.incomplete" | "response.failed";
154
+ turnId: string;
149
155
  type: "error" | "retry";
150
156
  };
151
157
 
152
158
  type CodexResponseDecision =
153
159
  | "continue_no_tools"
154
160
  | "retry_original_input"
161
+ | "return_compaction_boundary"
155
162
  | "return_terminal"
156
163
  | "return_tool_use";
157
164
 
@@ -399,6 +406,22 @@ function accumulateUsage(previous: Usage, current: Usage): Usage {
399
406
  };
400
407
  }
401
408
 
409
+ function reachedProviderCompactionThreshold(
410
+ scope: RuntimeScope | undefined,
411
+ usage: Usage,
412
+ model: Model<any>,
413
+ ): boolean {
414
+ const threshold = scope?.config.autoCompactAtPercent;
415
+ if (threshold === undefined || model.contextWindow <= 0 || usage.output >= model.maxTokens) {
416
+ return false;
417
+ }
418
+ const contextTokens =
419
+ usage.totalTokens > 0
420
+ ? usage.totalTokens
421
+ : usage.input + usage.output + usage.cacheRead + usage.cacheWrite;
422
+ return contextTokens > 0 && (contextTokens / model.contextWindow) * 100 >= threshold;
423
+ }
424
+
402
425
  function retryableResponseFailure(response: JsonRecord | undefined): boolean {
403
426
  const error = isObject(response?.["error"]) ? response["error"] : undefined;
404
427
  const code = typeof error?.["code"] === "string" ? error["code"].toLowerCase() : "";
@@ -412,6 +435,22 @@ function retryableResponseFailure(response: JsonRecord | undefined): boolean {
412
435
  );
413
436
  }
414
437
 
438
+ function terminalReason(terminalState: CodexTerminalState): string | undefined {
439
+ if (terminalState.type === "response.failed") {
440
+ const error = isObject(terminalState.response?.["error"])
441
+ ? terminalState.response["error"]
442
+ : undefined;
443
+ return typeof error?.["code"] === "string" ? error["code"] : undefined;
444
+ }
445
+ if (terminalState.type === "response.incomplete") {
446
+ const details = isObject(terminalState.response?.["incomplete_details"])
447
+ ? terminalState.response["incomplete_details"]
448
+ : undefined;
449
+ return typeof details?.["reason"] === "string" ? details["reason"] : undefined;
450
+ }
451
+ return undefined;
452
+ }
453
+
415
454
  function terminalErrorMessage(disposition: CodexPostToolDisposition): string {
416
455
  if (disposition.terminalType === "response.failed") {
417
456
  const error = isObject(disposition.response?.["error"])
@@ -590,6 +629,7 @@ export class CodexProviderRuntime {
590
629
  private readonly prewarmedTemplates = new Set<string>();
591
630
  private readonly requestTails = new Map<string, Promise<void>>();
592
631
  private readonly activeAgentTurns = new Map<string, ActiveAgentTurn>();
632
+ private readonly postToolDispositions = new Map<string, CodexPostToolDisposition>();
593
633
  private readonly windowNumbers = new Map<string, number>();
594
634
  private readonly activeThreadIds = new Map<string, string>();
595
635
  private readonly pi: ExtensionAPI;
@@ -638,7 +678,9 @@ export class CodexProviderRuntime {
638
678
  }
639
679
 
640
680
  beginAgentTurn(ctx: ExtensionContext): void {
641
- this.activeAgentTurns.set(ctx.sessionManager.getSessionId(), {
681
+ const sessionId = ctx.sessionManager.getSessionId();
682
+ this.clearPostToolDispositions((disposition) => disposition.sessionId === sessionId);
683
+ this.activeAgentTurns.set(sessionId, {
642
684
  turnId: uuidv7(),
643
685
  startedAtUnixMs: Date.now(),
644
686
  turnState: new CodexTurnState(),
@@ -646,7 +688,12 @@ export class CodexProviderRuntime {
646
688
  }
647
689
 
648
690
  endAgentTurn(ctx: ExtensionContext): void {
649
- this.activeAgentTurns.delete(ctx.sessionManager.getSessionId());
691
+ const sessionId = ctx.sessionManager.getSessionId();
692
+ const agentTurn = this.activeAgentTurns.get(sessionId);
693
+ if (agentTurn) {
694
+ this.clearPostToolDispositions((disposition) => disposition.turnId === agentTurn.turnId);
695
+ }
696
+ this.activeAgentTurns.delete(sessionId);
650
697
  }
651
698
 
652
699
  updateSessionConfig(sessionId: string, config: CodexCompatConfig): void {
@@ -671,6 +718,46 @@ export class CodexProviderRuntime {
671
718
  );
672
719
  }
673
720
 
721
+ private rememberPostToolDisposition(disposition: CodexPostToolDisposition): void {
722
+ const callIds = new Set(disposition.callIds);
723
+ if ([...callIds].some((callId) => this.postToolDispositions.has(callId))) {
724
+ throw new Error("Codex returned a tool call that already has a pending disposition.");
725
+ }
726
+ for (const callId of callIds) this.postToolDispositions.set(callId, disposition);
727
+ }
728
+
729
+ private findPostToolDisposition(context: Context): CodexPostToolDisposition | undefined {
730
+ const matches = new Set<CodexPostToolDisposition>();
731
+ for (const message of context.messages) {
732
+ if (message.role !== "assistant") continue;
733
+ for (const block of message.content) {
734
+ if (block.type !== "toolCall") continue;
735
+ const disposition = this.postToolDispositions.get(block.id);
736
+ if (disposition) matches.add(disposition);
737
+ }
738
+ }
739
+ if (matches.size > 1) {
740
+ throw new Error("Codex context contains multiple pending post-tool dispositions.");
741
+ }
742
+ return matches.values().next().value;
743
+ }
744
+
745
+ private forgetPostToolDisposition(disposition: CodexPostToolDisposition): void {
746
+ for (const callId of disposition.callIds) {
747
+ if (this.postToolDispositions.get(callId) === disposition) {
748
+ this.postToolDispositions.delete(callId);
749
+ }
750
+ }
751
+ }
752
+
753
+ private clearPostToolDispositions(
754
+ shouldClear: (disposition: CodexPostToolDisposition) => boolean,
755
+ ): void {
756
+ for (const [callId, disposition] of this.postToolDispositions) {
757
+ if (shouldClear(disposition)) this.postToolDispositions.delete(callId);
758
+ }
759
+ }
760
+
674
761
  private metadataIdentity(
675
762
  metadataSessionId: string | undefined,
676
763
  turn?: ActiveAgentTurn,
@@ -729,6 +816,7 @@ export class CodexProviderRuntime {
729
816
  this.clearPrewarmState(sessionId);
730
817
  this.requestTails.delete(sessionId);
731
818
  this.activeAgentTurns.delete(sessionId);
819
+ this.clearPostToolDispositions((disposition) => disposition.sessionId === sessionId);
732
820
  for (const key of this.windowNumbers.keys()) {
733
821
  if (key.startsWith(`${sessionId}\0`)) this.windowNumbers.delete(key);
734
822
  }
@@ -1153,6 +1241,7 @@ export class CodexProviderRuntime {
1153
1241
  };
1154
1242
  const runtimeSessionId = requestOptions.sessionId;
1155
1243
  let releaseRequest = () => {};
1244
+ let registeredPostToolDisposition: CodexPostToolDisposition | undefined;
1156
1245
  try {
1157
1246
  const accountId = validateCodexAuthentication(model, requestOptions.apiKey);
1158
1247
  releaseRequest = await this.acquireRequest(runtimeSessionId, requestOptions.signal);
@@ -1162,10 +1251,10 @@ export class CodexProviderRuntime {
1162
1251
  const agentTurn = this.agentTurn(runtimeSessionId);
1163
1252
  const responsesLiteEnabled = this.responsesLiteEnabled(runtimeSessionId);
1164
1253
  let carriedResponseRetries = 0;
1165
- const pendingPostToolDisposition = agentTurn.pendingPostToolDisposition;
1254
+ const pendingPostToolDisposition = this.findPostToolDisposition(context);
1166
1255
  if (pendingPostToolDisposition) {
1167
1256
  assertLinkedToolOutputs(context, pendingPostToolDisposition);
1168
- delete agentTurn.pendingPostToolDisposition;
1257
+ this.forgetPostToolDisposition(pendingPostToolDisposition);
1169
1258
  if (
1170
1259
  pendingPostToolDisposition.type === "error" ||
1171
1260
  pendingPostToolDisposition.retryAttempt > this.responseRetryPolicy.maxRetries
@@ -1229,6 +1318,7 @@ export class CodexProviderRuntime {
1229
1318
  );
1230
1319
 
1231
1320
  const rawItems: ResponsesItem[] = [];
1321
+ const nativeAttempts: NativeResponseAttempt[] = [];
1232
1322
  const prewarmDiagnostics: CodexTransportDiagnostic[] = [];
1233
1323
  await this.maybePrewarm({
1234
1324
  model,
@@ -1314,6 +1404,14 @@ export class CodexProviderRuntime {
1314
1404
  // `response.output_item.done` is Codex's item-level commit point. Terminal
1315
1405
  // response.output snapshots are deliberately ignored.
1316
1406
  const attemptItems = attemptCapture.streamedItems;
1407
+ if (terminalState.type) {
1408
+ const reason = terminalReason(terminalState);
1409
+ nativeAttempts.push({
1410
+ itemCount: attemptItems.length,
1411
+ terminalType: terminalState.type,
1412
+ ...(reason ? { terminalReason: reason } : {}),
1413
+ });
1414
+ }
1317
1415
  const toolCalls = assessAttemptToolCalls(attemptItems, attemptCapture);
1318
1416
  const incompleteDetails = isObject(terminalState.response?.["incomplete_details"])
1319
1417
  ? terminalState.response["incomplete_details"]
@@ -1356,15 +1454,18 @@ export class CodexProviderRuntime {
1356
1454
  terminalState.type === "response.incomplete" ||
1357
1455
  retryableResponseFailure(terminalState.response);
1358
1456
  postToolDisposition = retryable ? "retry" : "error";
1359
- agentTurn.pendingPostToolDisposition = {
1457
+ registeredPostToolDisposition = {
1360
1458
  callIds,
1361
1459
  ...(terminalState.response
1362
1460
  ? { response: structuredClone(terminalState.response) }
1363
1461
  : {}),
1364
1462
  retryAttempt: retryable ? responseRetries + 1 : responseRetries,
1463
+ ...(runtimeSessionId ? { sessionId: runtimeSessionId } : {}),
1365
1464
  terminalType: terminalState.type,
1465
+ turnId: agentTurn.turnId,
1366
1466
  type: postToolDisposition,
1367
1467
  };
1468
+ this.rememberPostToolDisposition(registeredPostToolDisposition);
1368
1469
  }
1369
1470
  discardIncompleteAttemptContent(output, attemptState);
1370
1471
  rawItems.push(...attemptItems.map((item) => structuredClone(item)));
@@ -1405,6 +1506,18 @@ export class CodexProviderRuntime {
1405
1506
  terminalState.type === "response.completed" &&
1406
1507
  terminalState.response?.["end_turn"] === false
1407
1508
  ) {
1509
+ const scope = runtimeSessionId ? this.scopes.get(runtimeSessionId) : undefined;
1510
+ if (reachedProviderCompactionThreshold(scope, output.usage, model)) {
1511
+ // Pi 0.84 has no provider event that can split one stream into two
1512
+ // assistant messages. Return the committed prefix as a recoverable
1513
+ // length boundary so Pi persists B1, runs native overflow compaction,
1514
+ // and continues from K without adding model-visible input.
1515
+ output.stopReason = "length";
1516
+ output.rawStopReason = PROVIDER_COMPACTION_BOUNDARY_STOP_REASON;
1517
+ delete output.errorMessage;
1518
+ recordDecision("return_compaction_boundary");
1519
+ break;
1520
+ }
1408
1521
  if (!nextBody) {
1409
1522
  throw new Error(
1410
1523
  "Codex requested a follow-up response, but its completed output could not be appended to request history.",
@@ -1460,7 +1573,7 @@ export class CodexProviderRuntime {
1460
1573
  if (!output.responseId) throw new Error("Codex response is missing a response id.");
1461
1574
  this.pi.appendEntry(
1462
1575
  NATIVE_RESPONSE_ENTRY_TYPE,
1463
- nativeResponseData(model.id, output.responseId, rawItems),
1576
+ nativeResponseData(model.id, output.responseId, rawItems, nativeAttempts),
1464
1577
  );
1465
1578
  }
1466
1579
  try {
@@ -1481,6 +1594,9 @@ export class CodexProviderRuntime {
1481
1594
  stream.push({ type: "done", reason: output.stopReason, message: output });
1482
1595
  stream.end();
1483
1596
  } catch (error) {
1597
+ if (registeredPostToolDisposition) {
1598
+ this.forgetPostToolDisposition(registeredPostToolDisposition);
1599
+ }
1484
1600
  clearStreamingScratchState(output);
1485
1601
  output.stopReason = requestOptions.signal?.aborted ? "aborted" : "error";
1486
1602
  output.errorMessage = formatProviderError(error);
@@ -14,7 +14,7 @@ import {
14
14
  isResponsesItem,
15
15
  type ResponsesItem,
16
16
  } from "./codex-protocol.ts";
17
- import { nativeResponseOverrides } from "./native-history.ts";
17
+ import { nativeCommittedPrefixBeforeOverflow, nativeResponseOverrides } from "./native-history.ts";
18
18
  import {
19
19
  CODEX_NAMESPACED_TOOL_NAMES,
20
20
  CODEX_TEXT_CONTENT_ITEM_TOOL_RESULT_NAMES,
@@ -279,10 +279,11 @@ export function providerHistory(options: {
279
279
  allTools: readonly ToolInfo[];
280
280
  grammarToolInputProperties?: GrammarToolInputProperties;
281
281
  imageDetail?: ImageDetail;
282
- dropLatestFailedAssistant?: boolean;
282
+ recoverLatestOverflowPrefix?: boolean;
283
283
  }): ResponsesItem[] {
284
284
  const branch = [...options.branch];
285
- if (options.dropLatestFailedAssistant) {
285
+ let recoveredPrefix: ResponsesItem[] = [];
286
+ if (options.recoverLatestOverflowPrefix) {
286
287
  const index = branch.findLastIndex(
287
288
  (entry) => entry.type === "message" && entry.message.role === "assistant",
288
289
  );
@@ -292,6 +293,14 @@ export function providerHistory(options: {
292
293
  entry.message.role === "assistant" &&
293
294
  (entry.message.stopReason === "error" || entry.message.stopReason === "aborted")
294
295
  ) {
296
+ if (entry.message.stopReason === "error" && entry.message.responseId) {
297
+ recoveredPrefix =
298
+ nativeCommittedPrefixBeforeOverflow(
299
+ branch,
300
+ options.wireModel.id,
301
+ entry.message.responseId,
302
+ ) ?? [];
303
+ }
295
304
  branch.splice(index, 1);
296
305
  }
297
306
  }
@@ -317,16 +326,20 @@ export function providerHistory(options: {
317
326
  options.imageDetail,
318
327
  nativeAssistantItems,
319
328
  ),
329
+ ...recoveredPrefix,
320
330
  ];
321
331
  }
322
332
 
323
333
  const context = buildSessionContext(branch);
324
- return encodeMessages(
325
- options.wireModel,
326
- convertToLlm(context.messages),
327
- options.allTools,
328
- options.grammarToolInputProperties ?? new Map(),
329
- options.imageDetail ?? "auto",
330
- nativeAssistantItems,
331
- );
334
+ return [
335
+ ...encodeMessages(
336
+ options.wireModel,
337
+ convertToLlm(context.messages),
338
+ options.allTools,
339
+ options.grammarToolInputProperties ?? new Map(),
340
+ options.imageDetail ?? "auto",
341
+ nativeAssistantItems,
342
+ ),
343
+ ...recoveredPrefix,
344
+ ];
332
345
  }
@@ -31,9 +31,9 @@ export interface CodexCompatConfig {
31
31
  /** Expose the standalone Codex web-search namespace tool. */
32
32
  webRun: boolean;
33
33
  /**
34
- * Compact at a provider request boundary when context usage reaches this
35
- * percentage. Omit it to rely only on Pi's compaction lifecycle (`/compact`,
36
- * threshold compaction, and overflow recovery).
34
+ * Compact at provider request boundaries when context usage reaches this
35
+ * percentage. Mid-response boundaries use Pi's bounded compact-and-continue
36
+ * lifecycle, so Pi auto-compaction must remain enabled.
37
37
  */
38
38
  autoCompactAtPercent?: number;
39
39
  webSearch: WebSearchMode;
@@ -3,6 +3,13 @@ import { isObject, isResponsesItem, type ResponsesItem } from "./codex-protocol.
3
3
 
4
4
  export const NATIVE_RESPONSE_ENTRY_TYPE = "openai-codex-compat-native-response";
5
5
  export const NATIVE_RESPONSE_FORMAT_VERSION = 1;
6
+ export const NATIVE_RESPONSE_ITEM_COMMIT = "response.output_item.done";
7
+
8
+ export type NativeResponseAttempt = {
9
+ itemCount: number;
10
+ terminalType: "response.completed" | "response.incomplete" | "response.failed";
11
+ terminalReason?: string;
12
+ };
6
13
 
7
14
  export type NativeResponseData = {
8
15
  kind: typeof NATIVE_RESPONSE_ENTRY_TYPE;
@@ -10,12 +17,15 @@ export type NativeResponseData = {
10
17
  modelId: string;
11
18
  responseId: string;
12
19
  items: ResponsesItem[];
20
+ itemCommit?: typeof NATIVE_RESPONSE_ITEM_COMMIT;
21
+ attempts?: NativeResponseAttempt[];
13
22
  };
14
23
 
15
24
  export function nativeResponseData(
16
25
  modelId: string,
17
26
  responseId: string,
18
27
  items: readonly ResponsesItem[],
28
+ attempts?: readonly NativeResponseAttempt[],
19
29
  ): NativeResponseData {
20
30
  return {
21
31
  kind: NATIVE_RESPONSE_ENTRY_TYPE,
@@ -23,6 +33,14 @@ export function nativeResponseData(
23
33
  modelId,
24
34
  responseId,
25
35
  items: items.map((item) => structuredClone(item)),
36
+ itemCommit: NATIVE_RESPONSE_ITEM_COMMIT,
37
+ ...(attempts
38
+ ? {
39
+ attempts: attempts.map((attempt) => ({
40
+ ...attempt,
41
+ })),
42
+ }
43
+ : {}),
26
44
  };
27
45
  }
28
46
 
@@ -45,15 +63,111 @@ export function parseNativeResponse(value: unknown): NativeResponseData | undefi
45
63
  }
46
64
  if (items.length === 0) return undefined;
47
65
 
66
+ const rawItemCommit = value["itemCommit"];
67
+ if (rawItemCommit !== undefined && rawItemCommit !== NATIVE_RESPONSE_ITEM_COMMIT) {
68
+ return undefined;
69
+ }
70
+ const rawAttempts = value["attempts"];
71
+ let attempts: NativeResponseAttempt[] | undefined;
72
+ if (rawAttempts !== undefined) {
73
+ if (!Array.isArray(rawAttempts)) return undefined;
74
+ attempts = [];
75
+ for (const rawAttempt of rawAttempts) {
76
+ if (
77
+ !isObject(rawAttempt) ||
78
+ !Number.isSafeInteger(rawAttempt["itemCount"]) ||
79
+ (rawAttempt["itemCount"] as number) < 0 ||
80
+ (rawAttempt["terminalType"] !== "response.completed" &&
81
+ rawAttempt["terminalType"] !== "response.incomplete" &&
82
+ rawAttempt["terminalType"] !== "response.failed") ||
83
+ (rawAttempt["terminalReason"] !== undefined &&
84
+ typeof rawAttempt["terminalReason"] !== "string")
85
+ ) {
86
+ return undefined;
87
+ }
88
+ attempts.push({
89
+ itemCount: rawAttempt["itemCount"] as number,
90
+ terminalType: rawAttempt["terminalType"],
91
+ ...(typeof rawAttempt["terminalReason"] === "string"
92
+ ? { terminalReason: rawAttempt["terminalReason"] }
93
+ : {}),
94
+ });
95
+ }
96
+ }
97
+
48
98
  return {
49
99
  kind: NATIVE_RESPONSE_ENTRY_TYPE,
50
100
  version: NATIVE_RESPONSE_FORMAT_VERSION,
51
101
  modelId: value.modelId,
52
102
  responseId: value["responseId"],
53
103
  items,
104
+ ...(rawItemCommit === NATIVE_RESPONSE_ITEM_COMMIT
105
+ ? { itemCommit: NATIVE_RESPONSE_ITEM_COMMIT }
106
+ : {}),
107
+ ...(attempts ? { attempts } : {}),
54
108
  };
55
109
  }
56
110
 
111
+ function linkedToolCalls(items: readonly ResponsesItem[]): boolean {
112
+ const unresolved = new Set<string>();
113
+ for (const item of items) {
114
+ if (item.type === "function_call" || item.type === "custom_tool_call") {
115
+ if (typeof item["call_id"] !== "string" || unresolved.has(item["call_id"])) return false;
116
+ unresolved.add(item["call_id"]);
117
+ continue;
118
+ }
119
+ if (item.type === "function_call_output" || item.type === "custom_tool_call_output") {
120
+ if (typeof item["call_id"] !== "string" || !unresolved.delete(item["call_id"])) {
121
+ return false;
122
+ }
123
+ }
124
+ }
125
+ return unresolved.size === 0;
126
+ }
127
+
128
+ /**
129
+ * Recover only done items from attempts completed before a final context-overflow
130
+ * subrequest. Older native entries remain replayable but lack enough provenance
131
+ * for this recovery path.
132
+ */
133
+ export function nativeCommittedPrefixBeforeOverflow(
134
+ branch: readonly SessionEntry[],
135
+ modelId: string,
136
+ responseId: string,
137
+ ): ResponsesItem[] | undefined {
138
+ for (let index = branch.length - 1; index >= 0; index--) {
139
+ const entry = branch[index]!;
140
+ if (entry.type !== "custom" || entry.customType !== NATIVE_RESPONSE_ENTRY_TYPE) continue;
141
+ const parsed = parseNativeResponse(entry.data);
142
+ if (!parsed) {
143
+ throw new Error(`Codex native response entry ${entry.id} is corrupt.`);
144
+ }
145
+ if (parsed.modelId !== modelId || parsed.responseId !== responseId) continue;
146
+ if (parsed.itemCommit !== NATIVE_RESPONSE_ITEM_COMMIT || !parsed.attempts) return undefined;
147
+ if (parsed.attempts.length < 2) return undefined;
148
+ if (
149
+ parsed.attempts.reduce((total, attempt) => total + attempt.itemCount, 0) !==
150
+ parsed.items.length
151
+ ) {
152
+ return undefined;
153
+ }
154
+
155
+ const finalAttempt = parsed.attempts.at(-1);
156
+ if (
157
+ finalAttempt?.terminalType !== "response.failed" ||
158
+ finalAttempt.terminalReason?.toLowerCase() !== "context_length_exceeded"
159
+ ) {
160
+ return undefined;
161
+ }
162
+ const prefixLength = parsed.items.length - finalAttempt.itemCount;
163
+ if (prefixLength <= 0) return undefined;
164
+ const prefix = parsed.items.slice(0, prefixLength);
165
+ if (!linkedToolCalls(prefix)) return undefined;
166
+ return prefix.map((item) => structuredClone(item));
167
+ }
168
+ return undefined;
169
+ }
170
+
57
171
  /** Load native assistant output overrides from the active Pi branch. */
58
172
  export function nativeResponseOverrides(
59
173
  branch: readonly SessionEntry[],
@@ -163,7 +163,7 @@ export default function registerRemoteCompaction(
163
163
  allTools,
164
164
  grammarToolInputProperties,
165
165
  imageDetail: config.imageDetail,
166
- dropLatestFailedAssistant: event.reason === "overflow" && event.willRetry,
166
+ recoverLatestOverflowPrefix: event.reason === "overflow" && event.willRetry,
167
167
  });
168
168
  const template =
169
169
  matching?.payload ??
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-openai-codex-compat",
3
- "version": "0.0.8",
3
+ "version": "0.0.9",
4
4
  "description": "OpenAI Codex compatibility for Pi with native compaction, fast mode, and Codex-optimized capabilities",
5
5
  "keywords": [
6
6
  "pi-package"