@gajae-code/agent-core 0.11.1 → 0.11.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/agent.ts CHANGED
@@ -308,6 +308,11 @@ interface CursorToolResultEntry {
308
308
  textLengthAtCall: number;
309
309
  }
310
310
 
311
+ export type AgentQueueSnapshot = {
312
+ steering: AgentMessage[];
313
+ followUp: AgentMessage[];
314
+ };
315
+
311
316
  export class Agent {
312
317
  #state: AgentState = {
313
318
  systemPrompt: [],
@@ -1009,6 +1014,20 @@ export class Agent {
1009
1014
  this.#followUpQueue = [...messages, ...this.#followUpQueue];
1010
1015
  }
1011
1016
 
1017
+ /** Snapshot both executable queues as one atomic session-level view. */
1018
+ snapshotQueues(): AgentQueueSnapshot {
1019
+ return {
1020
+ steering: this.#steeringQueue.slice(),
1021
+ followUp: this.#followUpQueue.slice(),
1022
+ };
1023
+ }
1024
+
1025
+ /** Replace both executable queues with a prior snapshot. */
1026
+ restoreQueues(snapshot: AgentQueueSnapshot): void {
1027
+ this.#steeringQueue = snapshot.steering.slice();
1028
+ this.#followUpQueue = snapshot.followUp.slice();
1029
+ }
1030
+
1012
1031
  #dequeueSteeringMessages(): AgentMessage[] {
1013
1032
  if (this.#steeringMode === "one-at-a-time") {
1014
1033
  if (this.#steeringQueue.length > 0) {
@@ -1607,7 +1626,7 @@ export class Agent {
1607
1626
  this.requestRunTerminal(managedLogicalRunOwner ?? runId, managedDecision.terminal);
1608
1627
  } else if (managedOutcome.type === "run_terminal") {
1609
1628
  this.requestRunTerminal(managedLogicalRunOwner ?? runId, { stopReason: managedOutcome.reason });
1610
- } else if (managedDecision?.type !== "retry") {
1629
+ } else if (managedDecision?.type !== "retry" && managedDecision?.type !== "maintenance") {
1611
1630
  this.#finalizeRun(managedLogicalRunOwner ?? runId);
1612
1631
  }
1613
1632
  }
@@ -1660,7 +1679,10 @@ export class Agent {
1660
1679
  });
1661
1680
  } finally {
1662
1681
  let continuation: ManagedAttemptContinuation | undefined;
1663
- if (managedOutcome?.type === "retryable_discarded" && managedDecision?.type === "retry") {
1682
+ if (
1683
+ managedOutcome?.type !== "run_terminal" &&
1684
+ (managedDecision?.type === "retry" || managedDecision?.type === "maintenance")
1685
+ ) {
1664
1686
  continuation = managedDecision.continuation;
1665
1687
  }
1666
1688
  const ownership: ManagedAttemptContinuationOwnership = {
@@ -1691,7 +1713,18 @@ export class Agent {
1691
1713
  if (continuation && ownership.isCurrent()) {
1692
1714
  try {
1693
1715
  await continuation(ownership);
1694
- if (this.#activeRunId === undefined && this.#managedLogicalRunOwner === managedLogicalRunOwner) {
1716
+ if (
1717
+ managedDecision?.type === "maintenance" &&
1718
+ this.#terminalizedLogicalRunIds.has(managedLogicalRunOwner ?? runId) &&
1719
+ this.#managedLogicalRunOwner === managedLogicalRunOwner
1720
+ ) {
1721
+ this.#managedLogicalRunOwner = undefined;
1722
+ }
1723
+ if (
1724
+ managedDecision?.type !== "maintenance" &&
1725
+ this.#activeRunId === undefined &&
1726
+ this.#managedLogicalRunOwner === managedLogicalRunOwner
1727
+ ) {
1695
1728
  this.#managedLogicalRunOwner = undefined;
1696
1729
  }
1697
1730
  } catch (err) {
@@ -29,7 +29,6 @@ import {
29
29
  withOpenAiRemoteCompactionPreserveData,
30
30
  } from "./openai";
31
31
  import autoHandoffThresholdFocusPrompt from "./prompts/auto-handoff-threshold-focus.md" with { type: "text" };
32
- import compactionShortSummaryPrompt from "./prompts/compaction-short-summary.md" with { type: "text" };
33
32
  import compactionSummaryPrompt from "./prompts/compaction-summary.md" with { type: "text" };
34
33
  import compactionTurnPrefixPrompt from "./prompts/compaction-turn-prefix.md" with { type: "text" };
35
34
  import compactionUpdateSummaryPrompt from "./prompts/compaction-update-summary.md" with { type: "text" };
@@ -510,7 +509,8 @@ function collectMessageFragments(message: AgentMessage): { fragments: string[];
510
509
  }
511
510
 
512
511
  switch (message.role) {
513
- case "user": {
512
+ case "user":
513
+ case "custom": {
514
514
  const content = (message as { content: string | Array<{ type: string; text?: string }> }).content;
515
515
  if (typeof content === "string") {
516
516
  fragments.push(content);
@@ -710,8 +710,6 @@ export function findCutPoint(
710
710
 
711
711
  for (let i = endIndex - 1; i >= startIndex; i--) {
712
712
  const entry = entries[i];
713
- if (entry.type !== "message") continue;
714
-
715
713
  // Estimate this message's size
716
714
  const messageTokens = estimateEntryTokens(entry);
717
715
  accumulatedTokens += messageTokens;
@@ -769,8 +767,6 @@ const SUMMARIZATION_PROMPT = prompt.render(compactionSummaryPrompt);
769
767
 
770
768
  const UPDATE_SUMMARIZATION_PROMPT = prompt.render(compactionUpdateSummaryPrompt);
771
769
 
772
- const SHORT_SUMMARY_PROMPT = prompt.render(compactionShortSummaryPrompt);
773
-
774
770
  const HANDOFF_DOCUMENT_PROMPT = prompt.render(handoffDocumentPrompt);
775
771
 
776
772
  export const AUTO_HANDOFF_THRESHOLD_FOCUS = prompt.render(autoHandoffThresholdFocusPrompt);
@@ -796,8 +792,7 @@ export interface SummaryOptions {
796
792
  /**
797
793
  * Optional telemetry handle. When provided, every LLM call emitted during
798
794
  * compaction is wrapped in an OTEL chat span tagged with
799
- * `pi.gen_ai.oneshot.kind` (`compaction_summary`, `compaction_short_summary`,
800
- * or `compaction_turn_prefix`). `undefined` keeps the call paths zero-cost.
795
+ * `pi.gen_ai.oneshot.kind` (`compaction_summary` or `compaction_turn_prefix`).
801
796
  */
802
797
  telemetry?: AgentTelemetry;
803
798
  authCredentialType?: "api_key" | "oauth";
@@ -1054,66 +1049,11 @@ export async function generateHandoff(
1054
1049
  .join("\n");
1055
1050
  }
1056
1051
 
1057
- async function generateShortSummary(
1058
- recentMessages: AgentMessage[],
1059
- historySummary: string | undefined,
1060
- model: Model,
1061
- reserveTokens: number,
1062
- apiKey: string,
1063
- signal?: AbortSignal,
1064
- options?: SummaryOptions,
1065
- ): Promise<string> {
1066
- const maxTokens = Math.min(512, Math.floor(0.2 * reserveTokens));
1067
- const llmMessages = (options?.convertToLlm ?? convertToLlm)(recentMessages);
1068
- const conversationText = boundConversationTextForSummary(serializeConversation(llmMessages), model, maxTokens);
1069
-
1070
- let promptText = `<conversation>\n${conversationText}\n</conversation>\n\n`;
1071
- if (historySummary) {
1072
- promptText += `<previous-summary>\n${historySummary}\n</previous-summary>\n\n`;
1073
- }
1074
- promptText += formatAdditionalContext(options?.extraContext);
1075
- promptText += SHORT_SUMMARY_PROMPT;
1076
-
1077
- if (options?.remoteEndpoint) {
1078
- const remote = await requestRemoteCompaction(
1079
- options.remoteEndpoint,
1080
- {
1081
- systemPrompt: SUMMARIZATION_SYSTEM_PROMPT,
1082
- prompt: promptText,
1083
- },
1084
- signal,
1085
- );
1086
- return remote.summary;
1087
- }
1088
-
1089
- const response = await instrumentedCompleteSimple(
1090
- model,
1091
- {
1092
- systemPrompt: [SUMMARIZATION_SYSTEM_PROMPT],
1093
- messages: [{ role: "user", content: [{ type: "text", text: promptText }], timestamp: Date.now() }],
1094
- },
1095
- {
1096
- maxTokens,
1097
- signal,
1098
- apiKey,
1099
- reasoning: Effort.High,
1100
- initiatorOverride: options?.initiatorOverride,
1101
- metadata: options?.metadata,
1102
- sessionId: options?.sessionId,
1103
- providerSessionState: options?.providerSessionState,
1104
- preferWebsockets: options?.preferWebsockets,
1105
- },
1106
- { telemetry: options?.telemetry, oneshotKind: "compaction_short_summary" },
1107
- );
1108
-
1109
- if (response.stopReason === "error") {
1110
- throw new Error(`Short summary failed: ${response.errorMessage || "Unknown error"}`);
1111
- }
1112
-
1113
- return response.content
1114
- .filter((c): c is { type: "text"; text: string } => c.type === "text")
1115
- .map(c => c.text)
1116
- .join("\n");
1052
+ /** Derive a display summary locally to avoid a second compaction LLM request. */
1053
+ function deriveShortSummary(summary: string): string {
1054
+ const firstParagraph = summary.trim().split(/\n\s*\n/, 1)[0] ?? "";
1055
+ const maxLength = 2_000;
1056
+ return firstParagraph.length <= maxLength ? firstParagraph : `${firstParagraph.slice(0, maxLength - 1)}…`;
1117
1057
  }
1118
1058
 
1119
1059
  // ============================================================================
@@ -1163,6 +1103,11 @@ export interface PrepareCompactionOptions {
1163
1103
  * (the confounded raw promptTokens/estimatedTokens quotient is never used).
1164
1104
  */
1165
1105
  tokenCorrectionRatio?: number;
1106
+ /**
1107
+ * Model context-window size. Windows below 66k retain the legacy fixed
1108
+ * keepRecentTokens behavior; larger windows scale the keep window to 30%.
1109
+ */
1110
+ contextWindow?: number;
1166
1111
  }
1167
1112
 
1168
1113
  export function prepareCompaction(
@@ -1193,13 +1138,42 @@ export function prepareCompaction(
1193
1138
  // counts system+tools+full history while estimatedTokens counted only the
1194
1139
  // post-boundary slice, so it was confounded and only ever shrank the window.
1195
1140
  // Here the correction is bidirectional and clamped to [0.5, 2].
1196
- const keepRecentTokens = settings.keepRecentTokens;
1141
+ const configuredKeepRecentTokens = settings.keepRecentTokens;
1142
+ const contextWindow = options.contextWindow;
1143
+ const thresholdSafeKeepRecentTokens =
1144
+ contextWindow !== undefined && Number.isFinite(contextWindow) && contextWindow > 1
1145
+ ? Math.max(
1146
+ 1,
1147
+ resolveThresholdTokens(contextWindow, settings) - effectiveReserveTokens(contextWindow, settings, 0),
1148
+ )
1149
+ : configuredKeepRecentTokens;
1150
+ const keepRecentTokens = Math.min(configuredKeepRecentTokens, thresholdSafeKeepRecentTokens);
1151
+ // Preserve the legacy fixed window for smaller models. At 66k and above,
1152
+ // retain up to 30% of the model context, but never enough to leave the
1153
+ // post-compaction prompt immediately above its configured threshold.
1154
+ const scaledKeepRecentTokens =
1155
+ contextWindow !== undefined && Number.isFinite(contextWindow) && contextWindow >= 66_000
1156
+ ? Math.min(thresholdSafeKeepRecentTokens, Math.max(keepRecentTokens, Math.floor(contextWindow * 0.3)))
1157
+ : keepRecentTokens;
1197
1158
  const rawRatio = options.tokenCorrectionRatio;
1198
1159
  const appliedRatio =
1199
1160
  rawRatio !== undefined && Number.isFinite(rawRatio) && rawRatio > 0
1200
1161
  ? Math.min(TOKEN_CORRECTION_MAX_RATIO, Math.max(TOKEN_CORRECTION_MIN_RATIO, rawRatio))
1201
1162
  : 1;
1202
- const keepRecentTokensCorrected = Math.max(1, Math.round(keepRecentTokens / appliedRatio));
1163
+ // Preserve an explicit keep floor that already covers the whole history: manual
1164
+ // and emergency callers rely on prepareCompaction returning undefined rather
1165
+ // than manufacturing a summary with no useful reduction. Otherwise, a scaled
1166
+ // window that exceeds a short history falls back to the threshold-safe floor.
1167
+ const historyTokens = pathEntries
1168
+ .slice(boundaryStart, boundaryEnd)
1169
+ .reduce((tokens, entry) => tokens + estimateEntryTokens(entry), 0);
1170
+ const effectiveKeepRecentTokens =
1171
+ configuredKeepRecentTokens > historyTokens
1172
+ ? configuredKeepRecentTokens
1173
+ : scaledKeepRecentTokens > keepRecentTokens && scaledKeepRecentTokens > historyTokens
1174
+ ? keepRecentTokens
1175
+ : scaledKeepRecentTokens;
1176
+ const keepRecentTokensCorrected = Math.max(1, Math.round(effectiveKeepRecentTokens / appliedRatio));
1203
1177
 
1204
1178
  const cutPoint = findCutPoint(pathEntries, boundaryStart, boundaryEnd, keepRecentTokensCorrected);
1205
1179
 
@@ -1437,28 +1411,10 @@ export async function compact(
1437
1411
  summary = "No prior history.";
1438
1412
  }
1439
1413
 
1440
- const shortSummary = await generateShortSummary(
1441
- recentMessages,
1442
- summary,
1443
- model,
1444
- settings.reserveTokens,
1445
- apiKey,
1446
- signal,
1447
- {
1448
- extraContext: options?.extraContext,
1449
- remoteEndpoint: summaryOptions.remoteEndpoint,
1450
- initiatorOverride: summaryOptions.initiatorOverride,
1451
- metadata: summaryOptions.metadata,
1452
- telemetry: summaryOptions.telemetry,
1453
- sessionId: summaryOptions.sessionId,
1454
- providerSessionState: summaryOptions.providerSessionState,
1455
- preferWebsockets: summaryOptions.preferWebsockets,
1456
- },
1457
- );
1458
-
1459
1414
  // Compute file lists and append to summary
1460
1415
  const { readFiles, modifiedFiles } = computeFileLists(fileOps);
1461
1416
  summary = upsertFileOperations(summary, readFiles, modifiedFiles);
1417
+ const shortSummary = deriveShortSummary(summary);
1462
1418
 
1463
1419
  if (!firstKeptEntryId) {
1464
1420
  throw new Error("First kept entry has no ID - session may need migration");
@@ -267,6 +267,56 @@ function readBasePath(path: string): string {
267
267
  return base;
268
268
  }
269
269
 
270
+ type ReadLineRange = { start: number; end: number };
271
+
272
+ const DEFAULT_READ_LINE_LIMIT = 500;
273
+
274
+ /** Parse trailing read selectors using the read tool's actual bounded default. */
275
+ function readLineRanges(path: string): ReadLineRange[] {
276
+ let target = path;
277
+ let raw = false;
278
+ while (/:(?:raw|conflicts)$/.test(target)) {
279
+ raw ||= target.endsWith(":raw");
280
+ target = target.replace(/:(?:raw|conflicts)$/, "");
281
+ }
282
+ const match = target.match(/:(\d+(?:[-+]\d+)?(?:,\d+(?:[-+]\d+)?)*)$/);
283
+ if (!match) return raw ? [{ start: 1, end: Number.POSITIVE_INFINITY }] : [];
284
+ return match[1].split(",").flatMap(part => {
285
+ const range = part.match(/^(\d+)(?:([-+])(\d+))?$/);
286
+ if (!range) return [];
287
+ const start = Number(range[1]);
288
+ const end =
289
+ range[2] === "+"
290
+ ? start + Number(range[3]) - 1
291
+ : range[2] === "-"
292
+ ? Number(range[3])
293
+ : start + DEFAULT_READ_LINE_LIMIT - 1;
294
+ return start > 0 && end >= start ? [{ start, end }] : [];
295
+ });
296
+ }
297
+
298
+ function strictlyContainsReadRange(container: ReadLineRange, contained: ReadLineRange): boolean {
299
+ return (
300
+ container.start <= contained.start &&
301
+ container.end >= contained.end &&
302
+ (container.start < contained.start || container.end > contained.end)
303
+ );
304
+ }
305
+
306
+ function readSupersedesRead(
307
+ later: ToolCall,
308
+ earlier: ToolCall,
309
+ lineRangesByCall: ReadonlyMap<ToolCall, ReadLineRange[]>,
310
+ ): boolean {
311
+ const laterRanges = lineRangesByCall.get(later);
312
+ const earlierRanges = lineRangesByCall.get(earlier);
313
+ return (
314
+ laterRanges?.length === 1 &&
315
+ earlierRanges?.length === 1 &&
316
+ strictlyContainsReadRange(laterRanges[0], earlierRanges[0])
317
+ );
318
+ }
319
+
270
320
  /**
271
321
  * Stable identity for "the same logical lookup": same tool re-targeting the
272
322
  * same subject. A later result with the same key supersedes earlier ones.
@@ -275,9 +325,23 @@ function readBasePath(path: string): string {
275
325
  * (`skip`) and result-shaping flags (`i`, `gitignore`): a later page or a
276
326
  * differently-shaped search complements earlier output, it does not replace it.
277
327
  */
328
+ const IDEMPOTENT_BASH_COMMAND =
329
+ /^(?:(?:bun|npm|pnpm|yarn)\s+(?:run\s+)?(?:test|build)\b|git\s+status\b|cargo\s+build\b|(?:make|just)\s+build\b)/;
330
+
331
+ function normalizedIdempotentBashCommand(call: ToolCall): string | undefined {
332
+ if (call.name !== "bash") return undefined;
333
+ const command = call.arguments.command;
334
+ if (typeof command !== "string") return undefined;
335
+ const normalized = command.trim().replace(/\s+/g, " ");
336
+ if (/[;&|]/.test(normalized) || !IDEMPOTENT_BASH_COMMAND.test(normalized)) return undefined;
337
+ return JSON.stringify([normalized, typeof call.arguments.cwd === "string" ? call.arguments.cwd : undefined]);
338
+ }
339
+
278
340
  function toolTargetKey(call: ToolCall): string | undefined {
279
341
  const path = toolCallPath(call);
280
342
  if (path !== undefined) return JSON.stringify([call.name, "path", path]);
343
+ const command = normalizedIdempotentBashCommand(call);
344
+ if (command !== undefined) return JSON.stringify([call.name, "command", command]);
281
345
  const pattern = call.arguments.pattern;
282
346
  if (typeof pattern === "string" && pattern.length > 0) {
283
347
  const paths = call.arguments.paths;
@@ -372,8 +436,9 @@ function buildStalenessIndex(entries: SessionEntry[]): StalenessIndex {
372
436
  }
373
437
  }
374
438
 
439
+ type ResultMeta = { key?: string; call: ToolCall; message: ToolResultMessage };
375
440
  const lastResultIndexByKey = new Map<string, number>();
376
- const resultMeta = new Map<number, { key?: string; call: ToolCall; message: ToolResultMessage }>();
441
+ const resultMeta = new Map<number, ResultMeta>();
377
442
  const lastEditIndexByPath = new Map<string, number>();
378
443
 
379
444
  for (let i = 0; i < entries.length; i++) {
@@ -439,6 +504,31 @@ function buildStalenessIndex(entries: SessionEntry[]): StalenessIndex {
439
504
  }
440
505
  }
441
506
 
507
+ const readsByBasePath = new Map<string, Array<[number, ResultMeta]>>();
508
+ const lineRangesByCall = new Map<ToolCall, ReadLineRange[]>();
509
+ for (const [index, meta] of resultMeta) {
510
+ if (meta.call.name !== "read") continue;
511
+ const path = toolCallPath(meta.call);
512
+ if (!path) continue;
513
+ lineRangesByCall.set(meta.call, readLineRanges(path));
514
+ const basePath = readBasePath(path);
515
+ const group = readsByBasePath.get(basePath);
516
+ if (group) group.push([index, meta]);
517
+ else readsByBasePath.set(basePath, [[index, meta]]);
518
+ }
519
+ for (const reads of readsByBasePath.values()) {
520
+ if (reads.length < 2) continue;
521
+ for (let earlier = 0; earlier < reads.length - 1; earlier++) {
522
+ const [index, meta] = reads[earlier];
523
+ for (let later = earlier + 1; later < reads.length; later++) {
524
+ if (readSupersedesRead(reads[later][1].call, meta.call, lineRangesByCall)) {
525
+ staleResultIndices.add(index);
526
+ break;
527
+ }
528
+ }
529
+ }
530
+ }
531
+
442
532
  return { staleResultIndices };
443
533
  }
444
534
  export function pruneAssistantToolArguments(
@@ -596,6 +686,13 @@ function collectToolOutputPruneCandidates(
596
686
  return { candidates, tokensSaved };
597
687
  }
598
688
 
689
+ function minimumSavings(config: PruneConfig, options: PruneToolOutputsOptions = {}): number {
690
+ const relaxedMinimum = options.relaxedMinimum;
691
+ return typeof relaxedMinimum === "number" && Number.isFinite(relaxedMinimum)
692
+ ? Math.min(config.minimumSavings, Math.max(0, relaxedMinimum))
693
+ : config.minimumSavings;
694
+ }
695
+
599
696
  /**
600
697
  * Estimate the token savings {@link pruneToolOutputs} would achieve, without
601
698
  * mutating any entry. Returns 0 savings when below the configured minimum so the
@@ -604,9 +701,10 @@ function collectToolOutputPruneCandidates(
604
701
  export function estimateToolOutputPruneSavings(
605
702
  entries: SessionEntry[],
606
703
  config: PruneConfig = DEFAULT_PRUNE_CONFIG,
704
+ options: PruneToolOutputsOptions = {},
607
705
  ): { prunableCount: number; tokensSaved: number } {
608
706
  const { candidates, tokensSaved } = collectToolOutputPruneCandidates(entries, config);
609
- if (tokensSaved < config.minimumSavings || candidates.length === 0) {
707
+ if (tokensSaved < minimumSavings(config, options) || candidates.length === 0) {
610
708
  return { prunableCount: 0, tokensSaved: 0 };
611
709
  }
612
710
  return { prunableCount: candidates.length, tokensSaved };
@@ -630,10 +728,20 @@ export function shouldRunMaintenancePrune(args: {
630
728
  return args.estimatedSavings > args.cacheEpochResetCost;
631
729
  }
632
730
 
633
- export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig = DEFAULT_PRUNE_CONFIG): PruneResult {
731
+ export interface PruneToolOutputsOptions {
732
+ /** Lower the usual minimum only when the caller is already over its compaction threshold. */
733
+ relaxedMinimum?: number;
734
+ }
735
+
736
+ export function pruneToolOutputs(
737
+ entries: SessionEntry[],
738
+ config: PruneConfig = DEFAULT_PRUNE_CONFIG,
739
+ options: PruneToolOutputsOptions = {},
740
+ ): PruneResult {
634
741
  const { candidates, tokensSaved } = collectToolOutputPruneCandidates(entries, config);
742
+ const minimum = minimumSavings(config, options);
635
743
 
636
- if (tokensSaved < config.minimumSavings || candidates.length === 0) {
744
+ if (tokensSaved < minimum || candidates.length === 0) {
637
745
  return { prunedCount: 0, tokensSaved: 0, prunedEntries: [] };
638
746
  }
639
747
 
package/src/proxy.ts CHANGED
@@ -30,6 +30,33 @@ class ProxyMessageEventStream extends EventStream<AssistantMessageEvent, Assista
30
30
  }
31
31
  }
32
32
 
33
+ interface ReasoningBuffers {
34
+ summary: string;
35
+ raw: string;
36
+ }
37
+
38
+ const reasoningBuffers = new WeakMap<object, ReasoningBuffers>();
39
+
40
+ function materializeReasoningProvenance(
41
+ content: Extract<AssistantMessage["content"][number], { type: "thinking" }>,
42
+ ): void {
43
+ const buffers = reasoningBuffers.get(content);
44
+ if (!buffers) return;
45
+ const mutable = content as { provenance?: "summary" | "raw" | "mixed"; summaryText?: string; rawText?: string };
46
+ if (mutable.provenance === undefined) {
47
+ if (mutable.summaryText === undefined && buffers.summary) mutable.summaryText = buffers.summary;
48
+ if (mutable.rawText === undefined && buffers.raw) mutable.rawText = buffers.raw;
49
+ mutable.provenance =
50
+ buffers.summary && buffers.raw ? "mixed" : buffers.summary ? "summary" : buffers.raw ? "raw" : undefined;
51
+ }
52
+ // Finalized display string must exclude raw CoT when a summary exists (parity with
53
+ // the Responses/Codex decoders): summary/mixed -> summary only; raw-only -> raw. The
54
+ // raw text stays available separately via rawText for explicit consumers.
55
+ const effSummary = mutable.summaryText ?? buffers.summary;
56
+ const effRaw = mutable.rawText ?? buffers.raw;
57
+ content.thinking = mutable.provenance === "raw" ? effRaw : effSummary || effRaw;
58
+ }
59
+
33
60
  /**
34
61
  * Proxy event types - server sends these with partial field stripped to reduce bandwidth.
35
62
  */
@@ -41,6 +68,9 @@ export type ProxyAssistantMessageEvent =
41
68
  | { type: "thinking_start"; contentIndex: number }
42
69
  | { type: "thinking_delta"; contentIndex: number; delta: string }
43
70
  | { type: "thinking_end"; contentIndex: number; contentSignature?: string }
71
+ | { type: "reasoning_summary_start"; contentIndex: number }
72
+ | { type: "reasoning_summary_delta"; contentIndex: number; delta: string }
73
+ | { type: "reasoning_summary_end"; contentIndex: number; content?: string }
44
74
  | { type: "toolcall_start"; contentIndex: number; id: string; toolName: string }
45
75
  | { type: "toolcall_delta"; contentIndex: number; delta: string }
46
76
  | { type: "toolcall_end"; contentIndex: number }
@@ -238,14 +268,22 @@ function processProxyEvent(
238
268
  throw new Error("Received text_end for non-text content");
239
269
  }
240
270
 
241
- case "thinking_start":
242
- partial.content[proxyEvent.contentIndex] = { type: "thinking", thinking: "" };
271
+ case "thinking_start": {
272
+ const content = { type: "thinking", thinking: "" } as Extract<
273
+ AssistantMessage["content"][number],
274
+ { type: "thinking" }
275
+ >;
276
+ partial.content[proxyEvent.contentIndex] = content;
277
+ reasoningBuffers.set(content, { summary: "", raw: "" });
243
278
  return { type: "thinking_start", contentIndex: proxyEvent.contentIndex, partial };
279
+ }
244
280
 
245
281
  case "thinking_delta": {
246
282
  const content = partial.content[proxyEvent.contentIndex];
247
283
  if (content?.type === "thinking") {
248
284
  content.thinking += proxyEvent.delta;
285
+ const buffers = reasoningBuffers.get(content);
286
+ if (buffers) buffers.raw += proxyEvent.delta;
249
287
  return {
250
288
  type: "thinking_delta",
251
289
  contentIndex: proxyEvent.contentIndex,
@@ -256,10 +294,52 @@ function processProxyEvent(
256
294
  throw new Error("Received thinking_delta for non-thinking content");
257
295
  }
258
296
 
297
+ case "reasoning_summary_start":
298
+ return { type: "reasoning_summary_start", contentIndex: proxyEvent.contentIndex, partial };
299
+
300
+ case "reasoning_summary_delta": {
301
+ const content = partial.content[proxyEvent.contentIndex];
302
+ if (content?.type === "thinking") {
303
+ content.thinking += proxyEvent.delta;
304
+ const buffers = reasoningBuffers.get(content);
305
+ if (buffers) buffers.summary += proxyEvent.delta;
306
+ return {
307
+ type: "reasoning_summary_delta",
308
+ contentIndex: proxyEvent.contentIndex,
309
+ delta: proxyEvent.delta,
310
+ partial,
311
+ };
312
+ }
313
+ throw new Error("Received reasoning_summary_delta for non-thinking content");
314
+ }
315
+
316
+ case "reasoning_summary_end": {
317
+ const content = partial.content[proxyEvent.contentIndex];
318
+ if (content?.type === "thinking") {
319
+ const buffers = reasoningBuffers.get(content);
320
+ // Final-only summaries arrive with the text on the end event and no summary
321
+ // deltas, so the accumulated buffer is empty. Prefer the end event's content
322
+ // so the summary survives the proxy and materializes as provenance "summary".
323
+ if (buffers && !buffers.summary.trim() && proxyEvent.content) {
324
+ buffers.summary = proxyEvent.content;
325
+ if (!content.thinking.trim()) content.thinking = proxyEvent.content;
326
+ }
327
+ materializeReasoningProvenance(content);
328
+ return {
329
+ type: "reasoning_summary_end",
330
+ contentIndex: proxyEvent.contentIndex,
331
+ content: buffers?.summary || proxyEvent.content || "",
332
+ partial,
333
+ };
334
+ }
335
+ throw new Error("Received reasoning_summary_end for non-thinking content");
336
+ }
337
+
259
338
  case "thinking_end": {
260
339
  const content = partial.content[proxyEvent.contentIndex];
261
340
  if (content?.type === "thinking") {
262
341
  content.thinkingSignature = proxyEvent.contentSignature;
342
+ materializeReasoningProvenance(content);
263
343
  return {
264
344
  type: "thinking_end",
265
345
  contentIndex: proxyEvent.contentIndex,
package/src/types.ts CHANGED
@@ -13,6 +13,7 @@ import type {
13
13
  Tool,
14
14
  ToolChoice,
15
15
  ToolResultMessage,
16
+ TransportFailureFacts,
16
17
  TSchema,
17
18
  } from "@gajae-code/ai";
18
19
  import type { AppendOnlyContextManager } from "./append-only-context";
@@ -58,6 +59,7 @@ export type ManagedAttemptContinuation = (ownership: ManagedAttemptContinuationO
58
59
  /** Decision returned by managed fallback policy for one provisional attempt. */
59
60
  export type ManagedAttemptDecision =
60
61
  | { type: "retry"; continuation: ManagedAttemptContinuation }
62
+ | { type: "maintenance"; continuation: ManagedAttemptContinuation }
61
63
  | { type: "terminal"; terminal: RunTerminalRequest };
62
64
 
63
65
  /** Structured result for one managed upstream invocation. */
@@ -67,9 +69,10 @@ export type ManagedAttemptOutcome =
67
69
  failure: {
68
70
  message: AssistantMessage;
69
71
  /** Exact provider transport facts, including retry headers, for fallback policy. */
70
- transportFailure?: import("@gajae-code/ai").TransportFailureFacts;
72
+ transportFailure?: TransportFailureFacts;
71
73
  };
72
74
  }
75
+ | { type: "context_overflow_discarded"; message: AssistantMessage }
73
76
  | { type: "run_terminal"; reason: "cancelled" | "error" | "exhausted" };
74
77
 
75
78
  export type ManagedAttemptOutcomeHandler = (