@oh-my-pi/pi-agent-core 18.2.0 → 18.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,21 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.2.1] - 2026-09-15
6
+
7
+ ### Added
8
+
9
+ - Added optional queued-message preparation with cancellation-safe delivery and appended context ([#11835](https://github.com/can1357/oh-my-pi/pull/11835) by [@andrebrait](https://github.com/andrebrait)).
10
+
11
+ ### Fixed
12
+
13
+ - Fixed streaming CPU blowup on long turns: per-delta `message_update` snapshots now deep-clone only the blocks the stream actually touched instead of the entire accumulated message, eliminating the quadratic cloning work that could freeze the TUI for tens of seconds to minutes while a subagent streams ([#10605](https://github.com/can1357/oh-my-pi/issues/10605)).
14
+ - Native compaction now carries an existing local summary into the first provider-native request instead of losing the summarized history. ([#11525](https://github.com/can1357/oh-my-pi/pull/11525) by [@rpie9](https://github.com/rpie9))
15
+ - Subsequent native compactions preserve messages appended between a speculative snapshot and its commit, while honoring `/clear` boundaries. ([#11525](https://github.com/can1357/oh-my-pi/pull/11525) by [@rpie9](https://github.com/rpie9))
16
+ - Native replay compatibility checks the active provider and Responses API independently of whether future native compaction is enabled. ([#11525](https://github.com/can1357/oh-my-pi/pull/11525) by [@rpie9](https://github.com/rpie9))
17
+ - Fixed compaction retaining oversized older steps beyond the recent-history budget and skipping previously retained history on later passes, preventing long tool loops from freeing enough context ([#11365](https://github.com/can1357/oh-my-pi/issues/11365)).
18
+ - Fixed Codex remote compaction retries for both Bun and proxy socket-closure messages and stopped falling back to the unsupported `/responses/compact` endpoint after V2 failures.
19
+
5
20
  ## [18.1.19] - 2026-09-12
6
21
 
7
22
  ### Added
@@ -3,7 +3,7 @@ import type { Dialect } from "@oh-my-pi/pi-ai/dialect";
3
3
  import type { HarmonyAuditEvent } from "@oh-my-pi/pi-ai/utils/harmony-leak";
4
4
  import type { AppendOnlyContextManager } from "./append-only-context.js";
5
5
  import { Tokenizer } from "./tokenizer.js";
6
- import type { AgentBeforeModelCall, AgentEvent, AgentLoopConfig, AgentMessage, AgentState, AgentTool, AgentToolContext, AgentTurnEndContext, AsideMessage, SpeculativeToolExecutionConfig, StreamFn, ToolCallContext, ToolChoiceDirective } from "./types.js";
6
+ import type { AgentBeforeModelCall, AgentEvent, AgentLoopConfig, AgentMessage, AgentState, AgentTool, AgentToolContext, AgentTurnEndContext, AsideMessage, PrepareQueuedMessages, SpeculativeToolExecutionConfig, StreamFn, ToolCallContext, ToolChoiceDirective } from "./types.js";
7
7
  export declare class AgentBusyError extends Error {
8
8
  constructor(message?: string);
9
9
  }
@@ -235,6 +235,8 @@ export declare class Agent {
235
235
  #private;
236
236
  streamFn: StreamFn;
237
237
  getApiKey?: (model: Model) => Promise<ApiKey | undefined> | ApiKey | undefined;
238
+ /** Prepare actual queue deliveries after dequeue gates; commit runs only while ownership remains valid. */
239
+ prepareQueuedMessages?: PrepareQueuedMessages;
238
240
  /**
239
241
  * Hook invoked after tool arguments are validated and before execution.
240
242
  * Reassign at any time to swap the implementation (e.g. on extension reload).
@@ -452,7 +454,8 @@ export declare class Agent {
452
454
  /** Non-consuming view of the pending steering queue (insertion order, newest
453
455
  * last). The session layer derives its queued-message display/count from
454
456
  * this live view instead of a mirror, so the agent-core queue stays the
455
- * single source of truth. */
457
+ * single source of truth. Includes exclusively claimed originals while
458
+ * preparation is pending, so editor restoration can cancel their delivery. */
456
459
  peekSteeringQueue(): readonly AgentMessage[];
457
460
  /** Non-consuming view of the pending follow-up queue. See
458
461
  * {@link peekSteeringQueue}. */
@@ -139,10 +139,11 @@ export interface CutPointResult {
139
139
  isSplitTurn: boolean;
140
140
  }
141
141
  /**
142
- * Find the cut point in session entries that keeps approximately `keepRecentTokens`.
142
+ * Find the oldest complete recent-history suffix that fits `keepRecentTokens`.
143
143
  *
144
- * Algorithm: Walk backwards from newest, accumulating estimated message sizes.
145
- * Stop when we've accumulated >= keepRecentTokens. Cut at that point.
144
+ * Walk backwards by valid cut points, measuring whole assistant/tool groups.
145
+ * Keep the newest group even when it alone exceeds the budget; never retain
146
+ * an additional older group that would push an otherwise fitting suffix over.
146
147
  *
147
148
  * Can cut at user OR assistant messages (never tool results). When cutting at an
148
149
  * assistant message with tool calls, its tool results come after and will be kept.
@@ -304,24 +305,25 @@ export interface CompactionPreparation {
304
305
  settings: CompactionSettings;
305
306
  }
306
307
  /**
307
- * Whether a prior remote compaction's provider-native replay can still be read
308
- * by the active model — the model that assembles the request context on every
309
- * turn. A local compaction (no remote preserve) always can: it holds a real
310
- * textual summary. A remote compaction (V2 or V1) only can when the active model
311
- * shares the blob's provider AND remote replay is still enabled; otherwise the
312
- * active model's encoder drops the payload (see `getOpenAIResponsesHistoryPayload`)
313
- * and only the opaque placeholder summary survives, so the caller must re-expand
314
- * the originals into a portable local summary rather than strand that history.
308
+ * Whether the active model's normal encoder can consume stored native history.
309
+ * Creating future compactions is a separate policy: disabling it does not disable
310
+ * Responses-family replay. Local summaries have no provider restriction.
311
+ */
312
+ export declare function canReplayRemoteCompaction(preserveData: Record<string, unknown> | undefined, activeModel: Model): boolean;
313
+ /**
314
+ * Whether compaction preparation may reuse a native boundary instead of
315
+ * re-expanding its original messages. This is deliberately stricter than normal
316
+ * replay: the active model must both read the payload and remain eligible for
317
+ * native compaction under the current settings. Otherwise local summarization
318
+ * needs the originals, not an opaque placeholder.
315
319
  *
316
- * Judged against the ACTIVE model, not the compaction candidate set: a role
317
- * model (e.g. `modelRoles.smol`) that still maps to the blob's provider does not
318
- * let the active model replay it, so keying reuse on "any candidate shares the
319
- * provider" left a provider-switched session permanently context-less (#6343).
320
+ * Main-session preparation is judged against the active model, not any role
321
+ * candidate, so a provider switch cannot strand the original history (#6343).
320
322
  */
321
323
  export declare function remotePreserveReusable(preserveData: Record<string, unknown> | undefined, activeModel: Model, settings: CompactionSettings): boolean;
322
324
  /**
323
- * Index of the newest compaction entry the active model can actually read, or
324
- * `-1` when none can.
325
+ * Index of the newest compaction boundary reusable under preparation policy,
326
+ * or `-1` when none can be reused (see {@link remotePreserveReusable}).
325
327
  *
326
328
  * A provider-native remote compaction (V2 or V1) stores an opaque replay payload
327
329
  * and only a placeholder summary, so for any OTHER provider that entry
@@ -31,6 +31,8 @@ export interface CompactionEntry<T = unknown> extends SessionEntryBase {
31
31
  shortSummary?: string;
32
32
  firstKeptEntryId: string;
33
33
  tokensBefore: number;
34
+ /** Last entry covered by native replay; later entries may precede the compaction record. */
35
+ providerReplayThroughEntryId?: string;
34
36
  /** Extension-specific data (e.g., ArtifactIndex, version markers for structured compaction) */
35
37
  details?: T;
36
38
  /** Hook-provided data to persist across compaction */
@@ -14,7 +14,7 @@
14
14
  * summarization endpoints that accept `{ systemPrompt, prompt }` and reply
15
15
  * with `{ summary, shortSummary? }`.
16
16
  */
17
- import type { CodexCompactionContext, FetchImpl, Message, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
17
+ import type { Api, CodexCompactionContext, FetchImpl, Message, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
18
18
  import { Tokenizer } from "../tokenizer.js";
19
19
  export * from "./compaction-v2-streaming.js";
20
20
  export declare const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
@@ -72,6 +72,7 @@ export interface RemoteCompactionResponse {
72
72
  summary: string;
73
73
  shortSummary?: string;
74
74
  }
75
+ export declare function isOpenAiRemoteCompactionApi(api: Api | undefined): boolean;
75
76
  export declare function shouldUseOpenAiRemoteCompaction(model: Model): boolean;
76
77
  export declare function getPreservedOpenAiRemoteCompactionData(preserveData: Record<string, unknown> | undefined): OpenAiRemoteCompactionPreserveData | undefined;
77
78
  export declare function withOpenAiRemoteCompactionPreserveData(preserveData: Record<string, unknown> | undefined, remoteCompaction: OpenAiRemoteCompactionPreserveData | undefined): Record<string, unknown> | undefined;
@@ -1,4 +1,4 @@
1
- import { Effort } from "@oh-my-pi/pi-ai";
1
+ import { Effort } from "@oh-my-pi/pi-catalog/effort";
2
2
  /**
3
3
  * Agent-local thinking selector.
4
4
  *
@@ -6,6 +6,13 @@ import type { AgentRunCoverage, AgentRunSummary } from "./run-collector.js";
6
6
  import type { AgentTelemetryConfig } from "./telemetry.js";
7
7
  /** Stream function - can return sync or Promise for async config lookup */
8
8
  export type StreamFn = (...args: Parameters<typeof streamSimple>) => AssistantMessageEventStream | Promise<AssistantMessageEventStream>;
9
+ /** Staged queue preparation; commit synchronously only while the batch is still owned. */
10
+ export interface QueuedMessagePreparation {
11
+ /** Append context after the originals; undefined stops this attempt, retaining originals unless explicitly removed. */
12
+ commit(): readonly AgentMessage[] | undefined;
13
+ }
14
+ /** Prepare an exclusively claimed batch. Undefined delivers unchanged; the signal also aborts when the claim is cancelled. */
15
+ export type PrepareQueuedMessages = (messages: readonly AgentMessage[], signal: AbortSignal) => QueuedMessagePreparation | undefined | Promise<QueuedMessagePreparation | undefined>;
9
16
  /** Called once an aside has been inserted into the agent's live context. */
10
17
  export declare const ASIDE_MESSAGE_COMMIT: unique symbol;
11
18
  /** Symbol-keyed handoff for one finalized, tool-owned stream speculation session. */
@@ -874,8 +881,9 @@ export interface AgentTool<TParameters extends TSchema = TSchema, TDetails = any
874
881
  * Called at `toolcall_start`, before any argument delta. Return `undefined` to opt out.
875
882
  */
876
883
  openArgStream?: (init: AgentToolArgStreamInit) => AgentToolArgStream | undefined;
877
- /** If true, tool is excluded unless explicitly listed in --tools or agent's tools field */
878
884
  hidden?: boolean;
885
+ /** If true, the tool can read `skill://<name>` instruction content; prompt builders gate skill guidance on it. */
886
+ readsSkillUris?: boolean;
879
887
  /** If true, tool can stage a pending action that requires explicit resolution via the resolve tool. */
880
888
  deferrable?: boolean;
881
889
  /** How an enabled tool is presented. See {@link ToolLoadMode}. Omitted is treated as `"essential"` for built-ins; custom-tool adapters normalize omission to `"discoverable"`. */
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-agent-core",
4
- "version": "18.2.0",
4
+ "version": "18.2.1",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Stencil Labs, Inc.",
@@ -35,16 +35,16 @@
35
35
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
36
36
  },
37
37
  "dependencies": {
38
- "@oh-my-pi/pi-ai": "18.2.0",
39
- "@oh-my-pi/pi-catalog": "18.2.0",
40
- "@oh-my-pi/pi-natives": "18.2.0",
41
- "@oh-my-pi/pi-utils": "18.2.0",
42
- "@oh-my-pi/pi-wire": "18.2.0",
43
- "@oh-my-pi/snapcompact": "18.2.0",
38
+ "@oh-my-pi/pi-ai": "18.2.1",
39
+ "@oh-my-pi/pi-catalog": "18.2.1",
40
+ "@oh-my-pi/pi-natives": "18.2.1",
41
+ "@oh-my-pi/pi-utils": "18.2.1",
42
+ "@oh-my-pi/pi-wire": "18.2.1",
43
+ "@oh-my-pi/snapcompact": "18.2.1",
44
44
  "@opentelemetry/api": "^1.9.1"
45
45
  },
46
46
  "devDependencies": {
47
- "@oh-my-pi/omptype": "18.2.0",
47
+ "@oh-my-pi/omptype": "18.2.1",
48
48
  "@opentelemetry/context-async-hooks": "^2.9.0",
49
49
  "@opentelemetry/sdk-trace-base": "^2.9.0",
50
50
  "@types/bun": "^1.3.14"
package/src/agent-loop.ts CHANGED
@@ -395,6 +395,64 @@ function snapshotAssistantMessage(message: AssistantMessage): AssistantMessage {
395
395
  };
396
396
  }
397
397
 
398
+ /**
399
+ * Incremental variant of `snapshotAssistantMessage` for per-delta
400
+ * `message_update` events.
401
+ *
402
+ * The stream contract guarantees that every content-block mutation a provider
403
+ * makes is paired with an event carrying that block's `contentIndex` (every
404
+ * provider mutates then pushes), and that `output.content` is append-only
405
+ * within a turn. A fresh snapshot therefore only needs to re-clone:
406
+ *
407
+ * - the block the current event targets (`changedIndex`),
408
+ * - blocks still open (started but not ended) — Cursor's edit block merges
409
+ * `path`/`stream_content` into a live block without an event, so open blocks
410
+ * are re-cloned on every delta,
411
+ * - blocks appended since the previous snapshot.
412
+ *
413
+ * Every other (finalized) block is carried over from the previous snapshot by
414
+ * reference: finalized blocks are never mutated after their end event, so
415
+ * sharing them is exact and turns the per-delta cost from O(turn content) into
416
+ * O(open blocks) — the difference between quadratic and linear streaming cost
417
+ * on long turns (issue #10605).
418
+ */
419
+ function snapshotAssistantMessageIncremental(
420
+ live: AssistantMessage,
421
+ prev: AssistantMessage,
422
+ changedIndex: number,
423
+ openBlocks: ReadonlySet<number>,
424
+ ): AssistantMessage {
425
+ const liveContent = live.content;
426
+ const prevContent = prev.content;
427
+ const prevLen = prevContent.length;
428
+ // Reference-copy the previous snapshot's block array (native-speed), then
429
+ // patch in fresh clones only where the live state moved: the block this
430
+ // delta targeted, blocks still streaming, and blocks appended since the
431
+ // previous snapshot. Finalized blocks keep their existing snapshot clone.
432
+ const content = prevContent.slice();
433
+ for (let i = prevLen; i < liveContent.length; i++) {
434
+ content.push(snapshotAssistantContentBlock(liveContent[i]!));
435
+ }
436
+ if (changedIndex >= 0 && changedIndex < prevLen && changedIndex < liveContent.length) {
437
+ content[changedIndex] = snapshotAssistantContentBlock(liveContent[changedIndex]!);
438
+ }
439
+ for (const openIndex of openBlocks) {
440
+ if (openIndex !== changedIndex && openIndex >= 0 && openIndex < prevLen && openIndex < liveContent.length) {
441
+ content[openIndex] = snapshotAssistantContentBlock(liveContent[openIndex]!);
442
+ }
443
+ }
444
+ return {
445
+ ...live,
446
+ content,
447
+ usage: {
448
+ ...live.usage,
449
+ cost: { ...live.usage.cost },
450
+ },
451
+ disabledFeatures: live.disabledFeatures ? [...live.disabledFeatures] : undefined,
452
+ toolCallAbortMessages: live.toolCallAbortMessages ? { ...live.toolCallAbortMessages } : undefined,
453
+ };
454
+ }
455
+
398
456
  /**
399
457
  * Deep-clone an assistant streaming event so subscribers get an immutable view.
400
458
  * Pass `partialSnapshot` when the caller has already snapshotted `event.partial`
@@ -1813,6 +1871,13 @@ async function streamAssistantResponse(
1813
1871
 
1814
1872
  let partialMessage: AssistantMessage | null = null;
1815
1873
  let addedPartial = false;
1874
+ // Previous `message_update` snapshot for the incremental rebuild below;
1875
+ // null until the turn's `start` event seeds it.
1876
+ let turnSnapshot: AssistantMessage | null = null;
1877
+ // Content indices of blocks that started streaming but have not ended
1878
+ // yet — re-cloned on every delta because live blocks may be patched
1879
+ // without a paired event (Cursor's silent edit-block merge).
1880
+ const openBlocks = new Set<number>();
1816
1881
  const completedToolCallIds = new Set<string>();
1817
1882
  const argStreams = new Map<number, { id: string; stream: AgentToolArgStream }>();
1818
1883
  const cancelArgStreams = (): void => {
@@ -2037,6 +2102,8 @@ async function streamAssistantResponse(
2037
2102
  // consumer treats both as read-only, so cloning the identical partial
2038
2103
  // twice per delta was pure waste.
2039
2104
  const messageSnapshot = snapshotAssistantMessage(partialMessage);
2105
+ turnSnapshot = messageSnapshot;
2106
+ openBlocks.clear();
2040
2107
  stream.push({
2041
2108
  type: "message_update",
2042
2109
  assistantMessageEvent: snapshotAssistantMessageEvent(event, messageSnapshot),
@@ -2045,7 +2112,8 @@ async function streamAssistantResponse(
2045
2112
  } else {
2046
2113
  context.messages.push(partialMessage);
2047
2114
  addedPartial = true;
2048
- stream.push({ type: "message_start", message: snapshotAssistantMessage(partialMessage) });
2115
+ turnSnapshot = snapshotAssistantMessage(partialMessage);
2116
+ stream.push({ type: "message_start", message: turnSnapshot });
2049
2117
  }
2050
2118
  break;
2051
2119
 
@@ -2127,11 +2195,34 @@ async function streamAssistantResponse(
2127
2195
  partialMessage = event.partial;
2128
2196
  context.messages[context.messages.length - 1] = partialMessage;
2129
2197
  config.onAssistantMessageEvent?.(partialMessage, event);
2130
- // `message` and `assistantMessageEvent.partial` intentionally share one
2131
- // immutable snapshot of the streaming partial: every message_update
2132
- // consumer treats both as read-only, so cloning the identical partial
2133
- // twice per delta was pure waste.
2134
- const messageSnapshot = snapshotAssistantMessage(partialMessage);
2198
+ // Track which blocks are still streaming: open blocks are
2199
+ // re-cloned on every delta, finalized blocks are shared.
2200
+ const contentIndex = (event as { contentIndex?: number }).contentIndex;
2201
+ if (contentIndex !== undefined) {
2202
+ if (event.type.endsWith("_start")) openBlocks.add(contentIndex);
2203
+ else if (event.type.endsWith("_end")) openBlocks.delete(contentIndex);
2204
+ }
2205
+ // READ-ONLY-CONSUMER INVARIANT: `message` and
2206
+ // `assistantMessageEvent.partial` intentionally share one snapshot,
2207
+ // and the snapshot is rebuilt incrementally — only the delta's
2208
+ // block, open blocks, and newly appended blocks are deep-cloned;
2209
+ // finalized blocks are carried over from the previous snapshot by
2210
+ // reference (see `snapshotAssistantMessageIncremental`), so they are
2211
+ // shared across ALL message_update snapshots of the turn.
2212
+ // Consumers MUST treat both fields — and every content block inside
2213
+ // them — as read-only: mutating a snapshot would corrupt every
2214
+ // earlier and later snapshot of the turn, not just this one. In
2215
+ // exchange, per-delta work is proportional to the live stream
2216
+ // instead of the whole turn (issue #10605).
2217
+ const messageSnapshot: AssistantMessage = turnSnapshot
2218
+ ? snapshotAssistantMessageIncremental(
2219
+ partialMessage,
2220
+ turnSnapshot,
2221
+ contentIndex ?? -1,
2222
+ openBlocks,
2223
+ )
2224
+ : snapshotAssistantMessage(partialMessage);
2225
+ turnSnapshot = messageSnapshot;
2135
2226
  stream.push({
2136
2227
  type: "message_update",
2137
2228
  assistantMessageEvent: snapshotAssistantMessageEvent(event, messageSnapshot),
package/src/agent.ts CHANGED
@@ -50,6 +50,7 @@ import type {
50
50
  AgentToolContext,
51
51
  AgentTurnEndContext,
52
52
  AsideMessage,
53
+ PrepareQueuedMessages,
53
54
  SpeculativeToolExecutionConfig,
54
55
  StreamFn,
55
56
  ToolCallContext,
@@ -362,6 +363,13 @@ interface CursorToolResultEntry {
362
363
  pending?: Promise<void>;
363
364
  }
364
365
 
366
+ type QueuedMessageQueue = "steering" | "followUp";
367
+
368
+ interface QueuedMessageClaim {
369
+ messages: AgentMessage[];
370
+ controller: AbortController;
371
+ }
372
+
365
373
  export class Agent {
366
374
  #state: AgentState = {
367
375
  systemPrompt: [],
@@ -383,6 +391,14 @@ export class Agent {
383
391
  #transformProviderContext?: (context: Context, model: Model) => Context | Promise<Context>;
384
392
  #steeringQueue: AgentMessage[] = [];
385
393
  #followUpQueue: AgentMessage[] = [];
394
+ #queuedMessageClaims: Partial<Record<QueuedMessageQueue, QueuedMessageClaim>> = {};
395
+ /** Dequeued originals remain recoverable until their transcript events arrive. */
396
+ #queuedMessageDeliveries = new Set<{
397
+ queue: QueuedMessageQueue;
398
+ controller: AbortController | undefined;
399
+ messages: AgentMessage[];
400
+ next: number;
401
+ }>();
386
402
  #steeringWaiters = new Set<() => void>();
387
403
 
388
404
  #steeringMode: "all" | "one-at-a-time";
@@ -449,6 +465,8 @@ export class Agent {
449
465
 
450
466
  streamFn: StreamFn;
451
467
  getApiKey?: (model: Model) => Promise<ApiKey | undefined> | ApiKey | undefined;
468
+ /** Prepare actual queue deliveries after dequeue gates; commit runs only while ownership remains valid. */
469
+ prepareQueuedMessages?: PrepareQueuedMessages;
452
470
  /**
453
471
  * Hook invoked after tool arguments are validated and before execution.
454
472
  * Reassign at any time to swap the implementation (e.g. on extension reload).
@@ -844,16 +862,90 @@ export class Agent {
844
862
  for (const hook of this.#beforeQueuedMessageDequeueHooks) await hook(signal);
845
863
  }
846
864
 
847
- async #dequeueSteeringMessagesAfterHooks(signal?: AbortSignal): Promise<AgentMessage[]> {
848
- if (signal?.aborted || this.#steeringQueue.length === 0) return [];
865
+ async #dequeueSteeringMessagesAfterHooks(signal: AbortSignal): Promise<AgentMessage[]> {
866
+ if (signal.aborted || this.#steeringQueue.length === 0) return [];
849
867
  await this.#runBeforeQueuedMessageDequeueHooks(signal);
850
- return signal?.aborted ? [] : this.#dequeueSteeringMessages();
868
+ return signal.aborted ? [] : this.#prepareQueuedMessageBatch("steering", signal);
851
869
  }
852
870
 
853
- async #dequeueFollowUpMessagesAfterHooks(signal?: AbortSignal): Promise<AgentMessage[]> {
854
- if (signal?.aborted || this.#followUpQueue.length === 0) return [];
871
+ async #dequeueFollowUpMessagesAfterHooks(signal: AbortSignal): Promise<AgentMessage[]> {
872
+ if (signal.aborted || this.#followUpQueue.length === 0) return [];
855
873
  await this.#runBeforeQueuedMessageDequeueHooks(signal);
856
- return signal?.aborted ? [] : this.#dequeueFollowUpMessages();
874
+ return signal.aborted ? [] : this.#prepareQueuedMessageBatch("followUp", signal);
875
+ }
876
+
877
+ async #prepareQueuedMessageBatch(queue: QueuedMessageQueue, signal: AbortSignal): Promise<AgentMessage[]> {
878
+ if (this.#queuedMessageClaims[queue]) return [];
879
+ const messages = queue === "steering" ? this.#dequeueSteeringMessages() : this.#dequeueFollowUpMessages();
880
+ const prepare = this.prepareQueuedMessages;
881
+ if (messages.length === 0) return messages;
882
+ const runController = this.#abortController;
883
+ if (!prepare) {
884
+ this.#queuedMessageDeliveries.add({ queue, controller: runController, messages, next: 0 });
885
+ return messages;
886
+ }
887
+
888
+ const claim: QueuedMessageClaim = { messages, controller: new AbortController() };
889
+ this.#queuedMessageClaims[queue] = claim;
890
+ const preparationSignal = AbortSignal.any([signal, claim.controller.signal]);
891
+ try {
892
+ const preparation = await prepare(messages, preparationSignal);
893
+ signal.throwIfAborted();
894
+ if (preparationSignal.aborted || this.#queuedMessageClaims[queue] !== claim) return [];
895
+ const additional = preparation?.commit();
896
+ if (preparationSignal.aborted || this.#queuedMessageClaims[queue] !== claim) return [];
897
+ if (preparation && additional === undefined) {
898
+ // Stop this attempt before the loop can immediately reclaim the restored batch.
899
+ runController?.abort();
900
+ throw new DOMException("Queued message preparation cancelled", "AbortError");
901
+ }
902
+ delete this.#queuedMessageClaims[queue];
903
+ this.#queuedMessageDeliveries.add({ queue, controller: runController, messages, next: 0 });
904
+ return additional?.length ? [...messages, ...additional] : messages;
905
+ } catch (error) {
906
+ if (signal.aborted) throw error;
907
+ if (preparationSignal.aborted || this.#queuedMessageClaims[queue] !== claim) return [];
908
+ throw error;
909
+ } finally {
910
+ if (this.#queuedMessageClaims[queue] === claim) this.#cancelQueuedMessagePreparation(queue, true);
911
+ }
912
+ }
913
+
914
+ #cancelQueuedMessagePreparation(queue: QueuedMessageQueue, restore = false): void {
915
+ if (!restore) {
916
+ for (const delivery of this.#queuedMessageDeliveries) {
917
+ if (delivery.queue === queue) this.#queuedMessageDeliveries.delete(delivery);
918
+ }
919
+ }
920
+ const claim = this.#queuedMessageClaims[queue];
921
+ if (!claim) return;
922
+ delete this.#queuedMessageClaims[queue];
923
+ if (restore) {
924
+ if (queue === "steering") {
925
+ this.#steeringQueue = [...claim.messages, ...this.#steeringQueue];
926
+ this.#notifySteeringWaiters();
927
+ } else {
928
+ this.#followUpQueue = [...claim.messages, ...this.#followUpQueue];
929
+ }
930
+ }
931
+ claim.controller.abort();
932
+ }
933
+
934
+ #restoreUndeliveredQueuedMessages(controller: AbortController): void {
935
+ if (this.#queuedMessageDeliveries.size === 0) return;
936
+ const restored: Record<QueuedMessageQueue, AgentMessage[]> = { steering: [], followUp: [] };
937
+ for (const delivery of this.#queuedMessageDeliveries) {
938
+ if (delivery.controller !== controller) continue;
939
+ this.#queuedMessageDeliveries.delete(delivery);
940
+ for (let i = delivery.next; i < delivery.messages.length; i++) {
941
+ restored[delivery.queue].push(delivery.messages[i]);
942
+ }
943
+ }
944
+ if (restored.steering.length > 0) {
945
+ this.#steeringQueue = [...restored.steering, ...this.#steeringQueue];
946
+ this.#notifySteeringWaiters();
947
+ }
948
+ if (restored.followUp.length > 0) this.#followUpQueue = [...restored.followUp, ...this.#followUpQueue];
857
949
  }
858
950
 
859
951
  setProviderResponseInterceptor(fn: SimpleStreamOptions["onResponse"] | undefined): void {
@@ -988,11 +1080,18 @@ export class Agent {
988
1080
  replaceQueues(steering: AgentMessage[], followUp: AgentMessage[]) {
989
1081
  this.#steeringQueue = steering.slice();
990
1082
  this.#followUpQueue = followUp.slice();
1083
+ this.#cancelQueuedMessagePreparation("steering");
1084
+ this.#cancelQueuedMessagePreparation("followUp");
991
1085
  this.#notifySteeringWaiters();
992
1086
  }
993
1087
 
994
1088
  appendMessage(m: AgentMessage) {
995
1089
  this.#state.messages.push(m);
1090
+ for (const delivery of this.#queuedMessageDeliveries) {
1091
+ if (delivery.messages[delivery.next] !== m) continue;
1092
+ if (++delivery.next === delivery.messages.length) this.#queuedMessageDeliveries.delete(delivery);
1093
+ break;
1094
+ }
996
1095
  }
997
1096
 
998
1097
  popMessage(): AgentMessage | undefined {
@@ -1022,11 +1121,13 @@ export class Agent {
1022
1121
 
1023
1122
  clearSteeringQueue() {
1024
1123
  this.#steeringQueue = [];
1124
+ this.#cancelQueuedMessagePreparation("steering");
1025
1125
  this.#notifySteeringWaiters();
1026
1126
  }
1027
1127
 
1028
1128
  clearFollowUpQueue() {
1029
1129
  this.#followUpQueue = [];
1130
+ this.#cancelQueuedMessagePreparation("followUp");
1030
1131
  }
1031
1132
 
1032
1133
  /**
@@ -1041,26 +1142,36 @@ export class Agent {
1041
1142
  clearAllQueues() {
1042
1143
  this.#steeringQueue = [];
1043
1144
  this.#followUpQueue = [];
1145
+ this.#cancelQueuedMessagePreparation("steering");
1146
+ this.#cancelQueuedMessagePreparation("followUp");
1044
1147
  this.#notifySteeringWaiters();
1045
1148
  this.clearDeferredToolDirectives();
1046
1149
  }
1047
1150
 
1048
1151
  hasQueuedMessages(): boolean {
1049
- return this.#steeringQueue.length > 0 || this.#followUpQueue.length > 0;
1152
+ return (
1153
+ this.#steeringQueue.length > 0 ||
1154
+ this.#followUpQueue.length > 0 ||
1155
+ this.#queuedMessageClaims.steering !== undefined ||
1156
+ this.#queuedMessageClaims.followUp !== undefined
1157
+ );
1050
1158
  }
1051
1159
 
1052
1160
  /** Non-consuming view of the pending steering queue (insertion order, newest
1053
1161
  * last). The session layer derives its queued-message display/count from
1054
1162
  * this live view instead of a mirror, so the agent-core queue stays the
1055
- * single source of truth. */
1163
+ * single source of truth. Includes exclusively claimed originals while
1164
+ * preparation is pending, so editor restoration can cancel their delivery. */
1056
1165
  peekSteeringQueue(): readonly AgentMessage[] {
1057
- return this.#steeringQueue;
1166
+ const claim = this.#queuedMessageClaims.steering;
1167
+ return claim ? [...claim.messages, ...this.#steeringQueue] : this.#steeringQueue;
1058
1168
  }
1059
1169
 
1060
1170
  /** Non-consuming view of the pending follow-up queue. See
1061
1171
  * {@link peekSteeringQueue}. */
1062
1172
  peekFollowUpQueue(): readonly AgentMessage[] {
1063
- return this.#followUpQueue;
1173
+ const claim = this.#queuedMessageClaims.followUp;
1174
+ return claim ? [...claim.messages, ...this.#followUpQueue] : this.#followUpQueue;
1064
1175
  }
1065
1176
 
1066
1177
  /** Nonblocking snapshot of the latest results, including provisional payloads while transforms are pending. */
@@ -1107,6 +1218,7 @@ export class Agent {
1107
1218
  * Used by dequeue keybinding.
1108
1219
  */
1109
1220
  popLastSteer(): AgentMessage | undefined {
1221
+ if (this.#steeringQueue.length === 0) this.#cancelQueuedMessagePreparation("steering", true);
1110
1222
  return this.#steeringQueue.pop();
1111
1223
  }
1112
1224
 
@@ -1115,6 +1227,7 @@ export class Agent {
1115
1227
  * Used by dequeue keybinding.
1116
1228
  */
1117
1229
  popLastFollowUp(): AgentMessage | undefined {
1230
+ if (this.#followUpQueue.length === 0) this.#cancelQueuedMessagePreparation("followUp", true);
1118
1231
  return this.#followUpQueue.pop();
1119
1232
  }
1120
1233
 
@@ -1154,15 +1267,19 @@ export class Agent {
1154
1267
  }
1155
1268
 
1156
1269
  reset() {
1270
+ if (this.#queuedMessageClaims.steering || this.#queuedMessageClaims.followUp) {
1271
+ this.#abortController?.abort();
1272
+ this.#abortController = undefined;
1273
+ this.#resolveRunningPrompt?.();
1274
+ this.#runningPrompt = undefined;
1275
+ this.#resolveRunningPrompt = undefined;
1276
+ }
1157
1277
  this.#state.messages.length = 0;
1158
1278
  this.#state.isStreaming = false;
1159
1279
  this.#state.streamMessage = null;
1160
1280
  this.#state.pendingToolCalls.clear();
1161
1281
  this.#state.error = undefined;
1162
- this.#steeringQueue = [];
1163
- this.#followUpQueue = [];
1164
- this.#notifySteeringWaiters();
1165
- this.clearDeferredToolDirectives();
1282
+ this.clearAllQueues();
1166
1283
  }
1167
1284
 
1168
1285
  /** Send a prompt with an AgentMessage */
@@ -1250,7 +1367,7 @@ export class Agent {
1250
1367
  this.#state.error = undefined;
1251
1368
 
1252
1369
  try {
1253
- const dequeueSignal = this.#continuationDequeueSignal(signal);
1370
+ const dequeueSignal = this.#continuationDequeueSignal(signal) ?? continuationAbortController.signal;
1254
1371
  const messages = this.#state.messages;
1255
1372
  if (messages.length === 0) {
1256
1373
  // An empty transcript has nothing to resume, but a queued steer/follow-up
@@ -1260,11 +1377,13 @@ export class Agent {
1260
1377
  // microtask because hasQueuedMessages() never clears, spinning an unbounded
1261
1378
  // allocation loop until OOM (issue #6344).
1262
1379
  const queuedSteering = await this.#dequeueSteeringMessagesAfterHooks(dequeueSignal);
1380
+ if (this.#abortController !== continuationAbortController) return;
1263
1381
  if (queuedSteering.length > 0) {
1264
1382
  await this.#runLoop(queuedSteering, { skipInitialSteeringPoll: true }, signal, true);
1265
1383
  return;
1266
1384
  }
1267
1385
  const queuedFollowUp = await this.#dequeueFollowUpMessagesAfterHooks(dequeueSignal);
1386
+ if (this.#abortController !== continuationAbortController) return;
1268
1387
  if (queuedFollowUp.length > 0) {
1269
1388
  await this.#runLoop(queuedFollowUp, undefined, signal, true);
1270
1389
  return;
@@ -1282,12 +1401,14 @@ export class Agent {
1282
1401
  return;
1283
1402
  }
1284
1403
  const queuedSteering = await this.#dequeueSteeringMessagesAfterHooks(dequeueSignal);
1404
+ if (this.#abortController !== continuationAbortController) return;
1285
1405
  if (queuedSteering.length > 0) {
1286
1406
  await this.#runLoop(queuedSteering, { skipInitialSteeringPoll: true }, signal, true);
1287
1407
  return;
1288
1408
  }
1289
1409
 
1290
1410
  const queuedFollowUp = await this.#dequeueFollowUpMessagesAfterHooks(dequeueSignal);
1411
+ if (this.#abortController !== continuationAbortController) return;
1291
1412
  if (queuedFollowUp.length > 0) {
1292
1413
  await this.#runLoop(queuedFollowUp, undefined, signal, true);
1293
1414
  return;
@@ -1298,6 +1419,7 @@ export class Agent {
1298
1419
 
1299
1420
  await this.#runLoop(undefined, undefined, signal, true);
1300
1421
  } finally {
1422
+ this.#restoreUndeliveredQueuedMessages(continuationAbortController);
1301
1423
  resolve();
1302
1424
  if (this.#abortController === continuationAbortController) {
1303
1425
  this.#state.isStreaming = false;
@@ -1513,7 +1635,7 @@ export class Agent {
1513
1635
  skipInitialSteeringPoll = false;
1514
1636
  return [];
1515
1637
  }
1516
- return this.#dequeueSteeringMessagesAfterHooks(signal);
1638
+ return this.#dequeueSteeringMessagesAfterHooks(signal ?? loopSignal);
1517
1639
  },
1518
1640
  hasSteeringMessages: () => {
1519
1641
  if (this.#steeringQueue.length === 0) {
@@ -1538,7 +1660,7 @@ export class Agent {
1538
1660
  },
1539
1661
  waitForSteeringMessages: signal => this.#waitForSteeringMessages(signal),
1540
1662
  hasIrcInterrupts: this.hasIrcInterrupts,
1541
- getFollowUpMessages: signal => this.#dequeueFollowUpMessagesAfterHooks(signal),
1663
+ getFollowUpMessages: signal => this.#dequeueFollowUpMessagesAfterHooks(signal ?? loopSignal),
1542
1664
  getAsideMessages: async () => (await this.#asideMessageProvider?.()) ?? [],
1543
1665
  onBeforeYield: () => this.#onBeforeYield?.(),
1544
1666
  telemetry: this.#telemetry,
@@ -1554,6 +1676,7 @@ export class Agent {
1554
1676
  : agentLoopContinue(context, config, loopSignal, this.streamFn);
1555
1677
 
1556
1678
  for await (const event of stream) {
1679
+ if (this.#abortController !== loopAbortController) return;
1557
1680
  if (event.type === "turn_start") turnOpen = true;
1558
1681
  if (event.type === "turn_end") turnOpen = false;
1559
1682
  // Update internal state based on events
@@ -1624,6 +1747,7 @@ export class Agent {
1624
1747
  }
1625
1748
  }
1626
1749
  } catch (err) {
1750
+ if (this.#abortController !== loopAbortController) return;
1627
1751
  const stoppedForAbort = loopSignal.aborted;
1628
1752
  const errorMessage = stoppedForAbort
1629
1753
  ? abortReasonText(loopSignal)
@@ -1728,6 +1852,7 @@ export class Agent {
1728
1852
  this.#emit({ type: "agent_end", messages: [errorMsg] });
1729
1853
  }
1730
1854
  } finally {
1855
+ this.#restoreUndeliveredQueuedMessages(loopAbortController);
1731
1856
  resolveRun?.();
1732
1857
  if (this.#abortController === loopAbortController) {
1733
1858
  this.#state.isStreaming = false;
@@ -32,7 +32,7 @@ import {
32
32
  OPENAI_HEADER_VALUES,
33
33
  OPENAI_HEADERS,
34
34
  } from "@oh-my-pi/pi-catalog/wire/codex";
35
- import { $env, logger, stringifyJson } from "@oh-my-pi/pi-utils";
35
+ import { $env, isUnexpectedSocketCloseMessage, logger, stringifyJson } from "@oh-my-pi/pi-utils";
36
36
 
37
37
  // ============================================================================
38
38
  // Types & Configuration
@@ -651,6 +651,7 @@ function isRetryableCompactionError(error: Error): boolean {
651
651
  }
652
652
  const message = error.message.toLowerCase();
653
653
  return (
654
+ isUnexpectedSocketCloseMessage(message) ||
654
655
  message.includes("stream closed before response.completed") ||
655
656
  message.includes("stream parse failed") ||
656
657
  message.includes("server_error") ||
@@ -72,6 +72,7 @@ import {
72
72
  import {
73
73
  buildOpenAiNativeHistory,
74
74
  getPreservedOpenAiRemoteCompactionData,
75
+ isOpenAiRemoteCompactionApi,
75
76
  requestOpenAiRemoteCompaction,
76
77
  requestRemoteCompaction,
77
78
  shouldUseOpenAiRemoteCompaction,
@@ -459,6 +460,17 @@ function findValidCutPoints(entries: SessionEntry[], startIndex: number, endInde
459
460
  return cutPoints;
460
461
  }
461
462
 
463
+ function isTurnStartEntry(entry: SessionEntry): boolean {
464
+ if (entry.type === "branch_summary" || entry.type === "custom_message") {
465
+ return true;
466
+ }
467
+ if (entry.type === "message") {
468
+ const role = entry.message.role as string;
469
+ return role === "user" || role === "bashExecution";
470
+ }
471
+ return false;
472
+ }
473
+
462
474
  /**
463
475
  * Find the user message (or bashExecution) that starts the turn containing the given entry index.
464
476
  * Returns -1 if no turn start found before the index.
@@ -466,17 +478,9 @@ function findValidCutPoints(entries: SessionEntry[], startIndex: number, endInde
466
478
  */
467
479
  export function findTurnStartIndex(entries: SessionEntry[], entryIndex: number, startIndex: number): number {
468
480
  for (let i = entryIndex; i >= startIndex; i--) {
469
- const entry = entries[i];
470
- // branch_summary and custom_message are user-role messages, can start a turn
471
- if (entry.type === "branch_summary" || entry.type === "custom_message") {
481
+ if (isTurnStartEntry(entries[i])) {
472
482
  return i;
473
483
  }
474
- if (entry.type === "message") {
475
- const role = entry.message.role as string;
476
- if (role === "user" || role === "bashExecution") {
477
- return i;
478
- }
479
- }
480
484
  }
481
485
  return -1;
482
486
  }
@@ -491,10 +495,11 @@ export interface CutPointResult {
491
495
  }
492
496
 
493
497
  /**
494
- * Find the cut point in session entries that keeps approximately `keepRecentTokens`.
498
+ * Find the oldest complete recent-history suffix that fits `keepRecentTokens`.
495
499
  *
496
- * Algorithm: Walk backwards from newest, accumulating estimated message sizes.
497
- * Stop when we've accumulated >= keepRecentTokens. Cut at that point.
500
+ * Walk backwards by valid cut points, measuring whole assistant/tool groups.
501
+ * Keep the newest group even when it alone exceeds the budget; never retain
502
+ * an additional older group that would push an otherwise fitting suffix over.
498
503
  *
499
504
  * Can cut at user OR assistant messages (never tool results). When cutting at an
500
505
  * assistant message with tool calls, its tool results come after and will be kept.
@@ -519,55 +524,45 @@ export function findCutPoint(
519
524
  return { firstKeptEntryIndex: startIndex, turnStartIndex: -1, isSplitTurn: false };
520
525
  }
521
526
 
522
- // Walk backwards from newest, accumulating estimated message sizes
527
+ // Evaluate the budget only at valid boundaries, after counting all results
528
+ // belonging to an assistant. Checking individual messages can either retain
529
+ // the oversized older assistant or miss its boundary and retain all history.
523
530
  let accumulatedTokens = 0;
524
- let cutIndex = cutPoints[0]; // Default: keep from first message (not header)
531
+ let cutPointIndex = cutPoints.length - 1;
532
+ let cutIndex = cutPoints[cutPointIndex];
525
533
 
526
534
  for (let i = endIndex - 1; i >= startIndex; i--) {
527
535
  const entry = entries[i];
528
- if (entry.type !== "message") continue;
529
-
530
- // Estimate this message's size
531
- const messageTokens = tokenizer.countMessage(entry.message);
532
- accumulatedTokens += messageTokens;
533
-
534
- // Check if we've exceeded the budget
535
- if (accumulatedTokens >= keepRecentTokens) {
536
- // Find the closest valid cut point at or after this entry
537
- for (let c = 0; c < cutPoints.length; c++) {
538
- if (cutPoints[c] >= i) {
539
- cutIndex = cutPoints[c];
540
- break;
541
- }
542
- }
543
- break;
544
- }
536
+ const message = getMessageFromEntry(entry);
537
+ if (message) accumulatedTokens += tokenizer.countMessage(message);
538
+ if (i !== cutPoints[cutPointIndex]) continue;
539
+ if (accumulatedTokens > keepRecentTokens) break;
540
+ cutIndex = i;
541
+ cutPointIndex--;
545
542
  }
546
543
 
547
- // Scan backwards from cutIndex to include any non-message entries (bash, settings, etc.)
544
+ const isTurnStart = isTurnStartEntry(entries[cutIndex]);
545
+ const turnStartIndex = isTurnStart ? -1 : findTurnStartIndex(entries, cutIndex, startIndex);
546
+
547
+ // Scan backwards from cutIndex to include any non-message entries (settings changes, etc.)
548
548
  while (cutIndex > startIndex) {
549
549
  const prevEntry = entries[cutIndex - 1];
550
- // Stop at session header or compaction boundaries
551
- if (prevEntry.type === "compaction") {
550
+ // Stop at session header, compaction, or reset boundaries
551
+ if (prevEntry.type === "compaction" || prevEntry.type === "reset_boundary") {
552
552
  break;
553
553
  }
554
- if (prevEntry.type === "message") {
555
- // Stop if we hit any message
554
+ if (getMessageFromEntry(prevEntry)) {
555
+ // Stop if we hit any entry that contributes a message
556
556
  break;
557
557
  }
558
- // Include this non-message entry (bash, settings change, etc.)
558
+ // Include this non-message entry (settings change, label, etc.)
559
559
  cutIndex--;
560
560
  }
561
561
 
562
- // Determine if this is a split turn
563
- const cutEntry = entries[cutIndex];
564
- const isUserMessage = cutEntry.type === "message" && cutEntry.message.role === "user";
565
- const turnStartIndex = isUserMessage ? -1 : findTurnStartIndex(entries, cutIndex, startIndex);
566
-
567
562
  return {
568
563
  firstKeptEntryIndex: cutIndex,
569
564
  turnStartIndex,
570
- isSplitTurn: !isUserMessage && turnStartIndex !== -1,
565
+ isSplitTurn: !isTurnStart && turnStartIndex !== -1,
571
566
  };
572
567
  }
573
568
 
@@ -1267,19 +1262,27 @@ export interface CompactionPreparation {
1267
1262
  }
1268
1263
 
1269
1264
  /**
1270
- * Whether a prior remote compaction's provider-native replay can still be read
1271
- * by the active model — the model that assembles the request context on every
1272
- * turn. A local compaction (no remote preserve) always can: it holds a real
1273
- * textual summary. A remote compaction (V2 or V1) only can when the active model
1274
- * shares the blob's provider AND remote replay is still enabled; otherwise the
1275
- * active model's encoder drops the payload (see `getOpenAIResponsesHistoryPayload`)
1276
- * and only the opaque placeholder summary survives, so the caller must re-expand
1277
- * the originals into a portable local summary rather than strand that history.
1265
+ * Whether the active model's normal encoder can consume stored native history.
1266
+ * Creating future compactions is a separate policy: disabling it does not disable
1267
+ * Responses-family replay. Local summaries have no provider restriction.
1268
+ */
1269
+ export function canReplayRemoteCompaction(
1270
+ preserveData: Record<string, unknown> | undefined,
1271
+ activeModel: Model,
1272
+ ): boolean {
1273
+ const remote = getCompactionV2PreserveData(preserveData) ?? getPreservedOpenAiRemoteCompactionData(preserveData);
1274
+ return !remote || (remote.provider === activeModel.provider && isOpenAiRemoteCompactionApi(activeModel.api));
1275
+ }
1276
+
1277
+ /**
1278
+ * Whether compaction preparation may reuse a native boundary instead of
1279
+ * re-expanding its original messages. This is deliberately stricter than normal
1280
+ * replay: the active model must both read the payload and remain eligible for
1281
+ * native compaction under the current settings. Otherwise local summarization
1282
+ * needs the originals, not an opaque placeholder.
1278
1283
  *
1279
- * Judged against the ACTIVE model, not the compaction candidate set: a role
1280
- * model (e.g. `modelRoles.smol`) that still maps to the blob's provider does not
1281
- * let the active model replay it, so keying reuse on "any candidate shares the
1282
- * provider" left a provider-switched session permanently context-less (#6343).
1284
+ * Main-session preparation is judged against the active model, not any role
1285
+ * candidate, so a provider switch cannot strand the original history (#6343).
1283
1286
  */
1284
1287
  export function remotePreserveReusable(
1285
1288
  preserveData: Record<string, unknown> | undefined,
@@ -1288,15 +1291,16 @@ export function remotePreserveReusable(
1288
1291
  ): boolean {
1289
1292
  const remote = getCompactionV2PreserveData(preserveData) ?? getPreservedOpenAiRemoteCompactionData(preserveData);
1290
1293
  if (!remote) return true;
1291
- if (settings.remoteEnabled === false) return false;
1292
- if (remote.provider !== activeModel.provider) return false;
1293
- const v2Ok = settings.remoteStreamingV2Enabled !== false && shouldUseCompactionV2Streaming(activeModel);
1294
- return v2Ok || shouldUseOpenAiRemoteCompaction(activeModel);
1294
+ return (
1295
+ remote.provider === activeModel.provider &&
1296
+ isOpenAiRemoteCompactionApi(activeModel.api) &&
1297
+ shouldUseProviderNativeCompaction(activeModel, settings)
1298
+ );
1295
1299
  }
1296
1300
 
1297
1301
  /**
1298
- * Index of the newest compaction entry the active model can actually read, or
1299
- * `-1` when none can.
1302
+ * Index of the newest compaction boundary reusable under preparation policy,
1303
+ * or `-1` when none can be reused (see {@link remotePreserveReusable}).
1300
1304
  *
1301
1305
  * A provider-native remote compaction (V2 or V1) stores an opaque replay payload
1302
1306
  * and only a placeholder summary, so for any OTHER provider that entry
@@ -1331,14 +1335,16 @@ export function prepareCompaction(
1331
1335
  activeModel?: Model,
1332
1336
  tokenizer: Tokenizer = new Tokenizer(activeModel),
1333
1337
  ): CompactionPreparation | undefined {
1334
- if (pathEntries.length > 0 && pathEntries[pathEntries.length - 1].type === "compaction") {
1338
+ const lastEntry = pathEntries[pathEntries.length - 1];
1339
+ // A speculative native record may leave uncovered messages before the record.
1340
+ if (lastEntry?.type === "compaction" && !lastEntry.providerReplayThroughEntryId) {
1335
1341
  return undefined;
1336
1342
  }
1337
1343
 
1338
1344
  let prevCompactionIndex = findReadableCompactionIndex(pathEntries, settings, activeModel);
1339
1345
 
1340
- // A reset after the reusable compaction clears its summary too. An older
1341
- // reset still bounds how far we may recover that compaction's kept messages.
1346
+ // A newer reset clears the previous summary. An older reset bounds both
1347
+ // the local retained tail and the native snapshot-to-commit interval.
1342
1348
  let resetBoundaryIndex = -1;
1343
1349
  for (let i = pathEntries.length - 1; i >= 0; i--) {
1344
1350
  if (pathEntries[i].type === "reset_boundary") {
@@ -1354,9 +1360,20 @@ export function prepareCompaction(
1354
1360
  let boundaryStart = Math.max(prevCompactionIndex, resetBoundaryIndex) + 1;
1355
1361
  if (
1356
1362
  previousCompaction &&
1357
- !getCompactionV2PreserveData(previousCompaction.preserveData) &&
1358
- !getPreservedOpenAiRemoteCompactionData(previousCompaction.preserveData)
1363
+ (getCompactionV2PreserveData(previousCompaction.preserveData) ||
1364
+ getPreservedOpenAiRemoteCompactionData(previousCompaction.preserveData))
1359
1365
  ) {
1366
+ if (previousCompaction.providerReplayThroughEntryId) {
1367
+ const replayThroughIndex = pathEntries.findIndex(
1368
+ entry => entry.id === previousCompaction.providerReplayThroughEntryId,
1369
+ );
1370
+ if (replayThroughIndex >= 0 && replayThroughIndex < prevCompactionIndex) {
1371
+ // Native replay covers the snapshot, not messages appended while the
1372
+ // request was running. Include that interval in the next preparation.
1373
+ boundaryStart = Math.max(replayThroughIndex, resetBoundaryIndex) + 1;
1374
+ }
1375
+ }
1376
+ } else if (previousCompaction) {
1360
1377
  // Local summaries exclude the retained tail, whose original entries precede
1361
1378
  // the compaction record. Native replay already carries that tail. Only look
1362
1379
  // backwards: advisor snapshots put all retained messages after the summary
@@ -1605,12 +1622,21 @@ export async function compact(
1605
1622
  const snapcompactArchiveMigrationMessage = previousSnapcompactArchiveText
1606
1623
  ? createSnapcompactArchiveMigrationMessage(previousSnapcompactArchiveText)
1607
1624
  : undefined;
1625
+ const previousNativeHistory =
1626
+ getCompactionV2PreserveData(previousPreserveData) ?? getPreservedOpenAiRemoteCompactionData(previousPreserveData);
1627
+ // A local summary has no native payload to carry it into the first remote
1628
+ // request. Encode it as history; do not resend opaque native placeholders.
1629
+ const previousSummaryMigrationMessage =
1630
+ settings.remoteEnabled !== false && previousSummary && !previousNativeHistory
1631
+ ? createCompactionSummaryMessage(previousSummary, tokensBefore, new Date().toISOString())
1632
+ : undefined;
1608
1633
 
1609
1634
  let preserveData = withAnthropicCompactionPreserveData(
1610
1635
  withOpenAiRemoteCompactionPreserveData(previousPreserveData, undefined),
1611
1636
  undefined,
1612
1637
  );
1613
1638
  const remoteMessages: AgentMessage[] = [
1639
+ ...(previousSummaryMigrationMessage ? [previousSummaryMigrationMessage] : []),
1614
1640
  ...(snapcompactArchiveMigrationMessage ? [snapcompactArchiveMigrationMessage] : []),
1615
1641
  ...messagesToSummarize,
1616
1642
  ...turnPrefixMessages,
@@ -37,6 +37,8 @@ export interface CompactionEntry<T = unknown> extends SessionEntryBase {
37
37
  shortSummary?: string;
38
38
  firstKeptEntryId: string;
39
39
  tokensBefore: number;
40
+ /** Last entry covered by native replay; later entries may precede the compaction record. */
41
+ providerReplayThroughEntryId?: string;
40
42
  /** Extension-specific data (e.g., ArtifactIndex, version markers for structured compaction) */
41
43
  details?: T;
42
44
  /** Hook-provided data to persist across compaction */
@@ -282,15 +282,22 @@ export interface RemoteCompactionResponse {
282
282
  // OpenAI provider gating + endpoint resolution
283
283
  // ============================================================================
284
284
 
285
- function isOpenAiRemoteCompactionApi(api: Api | undefined): boolean {
285
+ export function isOpenAiRemoteCompactionApi(api: Api | undefined): boolean {
286
286
  return api === "openai-responses" || api === "azure-openai-responses" || api === "openai-codex-responses";
287
287
  }
288
288
 
289
289
  export function shouldUseOpenAiRemoteCompaction(model: Model): boolean {
290
290
  if (model.remoteCompaction?.enabled === false) return false;
291
- if (model.provider === "openai" || model.provider === "openai-codex") return true;
291
+ const compactionApi = model.remoteCompaction?.api ?? model.api;
292
+ // ChatGPT's Codex backend exposes V2 compaction on /codex/responses, but
293
+ // does not expose the OpenAI V1 /responses/compact endpoint. Only use the
294
+ // V1 path for Codex when an explicit compatible endpoint was configured.
295
+ if (model.provider === "openai-codex") {
296
+ return (model.remoteCompaction?.endpoint?.trim().length ?? 0) > 0;
297
+ }
298
+ if (model.provider === "openai") return true;
292
299
  if (model.remoteCompaction?.enabled !== true) return false;
293
- return isOpenAiRemoteCompactionApi(model.remoteCompaction.api ?? model.api);
300
+ return isOpenAiRemoteCompactionApi(compactionApi);
294
301
  }
295
302
 
296
303
  function resolveOpenAiCompactEndpoint(model: Model): string {
package/src/telemetry.ts CHANGED
@@ -1756,6 +1756,7 @@ export async function instrumentedCompleteSimple<TApi extends Api>(
1756
1756
  const message = span.retry
1757
1757
  ? await retryTransientCompletion(runOnce, {
1758
1758
  ...span.retry,
1759
+ provider: model.provider,
1759
1760
  // Framework-owned: the caller must not be able to detach the
1760
1761
  // abort signal or the header source by passing them itself.
1761
1762
  signal: options.signal,
package/src/thinking.ts CHANGED
@@ -1,4 +1,4 @@
1
- import { Effort } from "@oh-my-pi/pi-ai";
1
+ import { Effort } from "@oh-my-pi/pi-catalog/effort";
2
2
 
3
3
  /**
4
4
  * Agent-local thinking selector.
package/src/tokenizer.ts CHANGED
@@ -274,6 +274,7 @@ export class Tokenizer {
274
274
  }
275
275
  break;
276
276
  }
277
+ case "custom":
277
278
  case "hookMessage":
278
279
  case "toolResult": {
279
280
  if (typeof message.content === "string") {
package/src/types.ts CHANGED
@@ -31,6 +31,18 @@ export type StreamFn = (
31
31
  ...args: Parameters<typeof streamSimple>
32
32
  ) => AssistantMessageEventStream | Promise<AssistantMessageEventStream>;
33
33
 
34
+ /** Staged queue preparation; commit synchronously only while the batch is still owned. */
35
+ export interface QueuedMessagePreparation {
36
+ /** Append context after the originals; undefined stops this attempt, retaining originals unless explicitly removed. */
37
+ commit(): readonly AgentMessage[] | undefined;
38
+ }
39
+
40
+ /** Prepare an exclusively claimed batch. Undefined delivers unchanged; the signal also aborts when the claim is cancelled. */
41
+ export type PrepareQueuedMessages = (
42
+ messages: readonly AgentMessage[],
43
+ signal: AbortSignal,
44
+ ) => QueuedMessagePreparation | undefined | Promise<QueuedMessagePreparation | undefined>;
45
+
34
46
  /** Called once an aside has been inserted into the agent's live context. */
35
47
  export const ASIDE_MESSAGE_COMMIT = Symbol("aside-message-commit");
36
48
  /** Symbol-keyed handoff for one finalized, tool-owned stream speculation session. */
@@ -1000,8 +1012,9 @@ export interface AgentTool<
1000
1012
  * Called at `toolcall_start`, before any argument delta. Return `undefined` to opt out.
1001
1013
  */
1002
1014
  openArgStream?: (init: AgentToolArgStreamInit) => AgentToolArgStream | undefined;
1003
- /** If true, tool is excluded unless explicitly listed in --tools or agent's tools field */
1004
1015
  hidden?: boolean;
1016
+ /** If true, the tool can read `skill://<name>` instruction content; prompt builders gate skill guidance on it. */
1017
+ readsSkillUris?: boolean;
1005
1018
  /** If true, tool can stage a pending action that requires explicit resolution via the resolve tool. */
1006
1019
  deferrable?: boolean;
1007
1020
  /** How an enabled tool is presented. See {@link ToolLoadMode}. Omitted is treated as `"essential"` for built-ins; custom-tool adapters normalize omission to `"discoverable"`. */