@oh-my-pi/pi-ai 18.1.6 → 18.1.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,22 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.1.8] - 2026-09-03
6
+
7
+ ### Added
8
+
9
+ - Added GPT-6 Astra support for preserving prompt caching when changing the thinking level during a conversation across the OpenAI and OpenAI Codex providers.
10
+
11
+ ### Changed
12
+
13
+ - Updated OpenAI Codex requests to improve routing by communicating the selected model and service tier across Responses, WebSocket, and remote-compaction requests.
14
+
15
+ ## [18.1.7] - 2026-09-03
16
+
17
+ ### Fixed
18
+
19
+ - Fixed DeepSeek-family Responses replay (e.g. opencode-go) rejecting a resumed thinking-mode turn with `400 The reasoning_text in the thinking mode must be passed back to the API` when compaction dropped the turn's reasoning; a non-empty placeholder is now synthesized instead of an empty `reasoning_text` ([#10690](https://github.com/can1357/oh-my-pi/issues/10690)).
20
+
5
21
  ## [18.1.6] - 2026-09-03
6
22
 
7
23
  ### Breaking Changes
@@ -0,0 +1,64 @@
1
+ /**
2
+ * Mid-conversation reasoning effort via `configuration_update` input items
3
+ * (GPT-6 Astra; `model.compat.supportsConfigurationUpdate`).
4
+ *
5
+ * The request-level `reasoning.effort` is pinned to the value of the session's
6
+ * first request so the cached prompt prefix survives an effort change. Each
7
+ * later change is carried as a `configuration_update` item inserted at the
8
+ * tail of the transcript — before the user message it takes effect on, or
9
+ * after the latest tool result when the level changes inside a tool loop — and
10
+ * replayed at that position on every subsequent request until another update
11
+ * overrides it. Mirrors the Anthropic provider's stable `output_config.effort`
12
+ * planning.
13
+ *
14
+ * Used by both the platform Responses provider and the Codex provider; the
15
+ * state lives in each provider's session state, keyed per conversation.
16
+ *
17
+ * Wire constraints (verified against the Codex backend): only `gpt-6-astra`
18
+ * accepts the item type, consecutive updates are rejected, and
19
+ * `/responses/compact` rejects histories containing them — compaction
20
+ * requests are built outside this planner and never carry the items.
21
+ */
22
+ /** `configuration_update` input item; only `reasoning.effort` is updatable. */
23
+ export interface ConfigurationUpdateItem {
24
+ type: "configuration_update";
25
+ reasoning: {
26
+ effort: string;
27
+ };
28
+ }
29
+ interface EffortTransition<TEffort extends string> {
30
+ /** Input-array position the item is spliced into (before `input[index]`). */
31
+ index: number;
32
+ /** Fingerprint of `input[index - 1]` at record time; a mismatch means the history was rewritten. */
33
+ anchor: string;
34
+ effort: TEffort;
35
+ }
36
+ /** Per-conversation effort baseline and recorded transitions. */
37
+ export interface OpenAIEffortControlState<TEffort extends string = string> {
38
+ baseEffort?: TEffort;
39
+ currentEffort?: TEffort;
40
+ transitions: EffortTransition<TEffort>[];
41
+ }
42
+ export declare function createOpenAIEffortControlState<TEffort extends string>(): OpenAIEffortControlState<TEffort>;
43
+ /**
44
+ * Fetch (or create) the control state for one conversation from a provider's
45
+ * bounded per-session map, refreshing its LRU slot.
46
+ */
47
+ export declare function getOpenAIEffortControlState<TEffort extends string>(states: Map<string, OpenAIEffortControlState<TEffort>>, key: string): OpenAIEffortControlState<TEffort>;
48
+ interface AnchorableItem {
49
+ type?: string | null;
50
+ role?: string;
51
+ id?: string | null;
52
+ status?: string | null;
53
+ }
54
+ /**
55
+ * Pin the request-level effort to the session baseline and splice pending
56
+ * `configuration_update` items into `input` (mutated in place).
57
+ *
58
+ * `input` is the freshly built transcript for this request, without any
59
+ * `configuration_update` items. `requested` is the wire effort the caller
60
+ * would otherwise send at the request level. Returns the effort to send at the
61
+ * request level (`requested` on the first request, the baseline afterwards).
62
+ */
63
+ export declare function planStableOpenAIEffort<TItem extends AnchorableItem, TEffort extends string>(state: OpenAIEffortControlState<TEffort>, input: Array<TItem | ConfigurationUpdateItem>, requested: TEffort): TEffort;
64
+ export {};
@@ -2918,7 +2918,7 @@ export interface ResponseInputImageContent {
2918
2918
  * `assistant` role are presumed to have been generated by the model in previous
2919
2919
  * interactions.
2920
2920
  */
2921
- export type ResponseInputItem = EasyInputMessage | ResponseInputItem.Message | ResponseOutputMessage | ResponseFileSearchToolCall | ResponseComputerToolCall | ResponseInputItem.ComputerCallOutput | ResponseFunctionWebSearch | ResponseFunctionToolCall | ResponseInputItem.FunctionCallOutput | ResponseInputItem.ToolSearchCall | ResponseToolSearchOutputItemParam | ResponseInputItem.AdditionalTools | ResponseReasoningItem | ResponseCompactionItemParam | ResponseInputItem.ImageGenerationCall | ResponseCodeInterpreterToolCall | ResponseInputItem.LocalShellCall | ResponseInputItem.LocalShellCallOutput | ResponseInputItem.ShellCall | ResponseInputItem.ShellCallOutput | ResponseInputItem.ApplyPatchCall | ResponseInputItem.ApplyPatchCallOutput | ResponseInputItem.McpListTools | ResponseInputItem.McpApprovalRequest | ResponseInputItem.McpApprovalResponse | ResponseInputItem.McpCall | ResponseCustomToolCallOutput | ResponseCustomToolCall | ResponseInputItem.CompactionTrigger | ResponseInputItem.ItemReference;
2921
+ export type ResponseInputItem = EasyInputMessage | ResponseInputItem.Message | ResponseOutputMessage | ResponseFileSearchToolCall | ResponseComputerToolCall | ResponseInputItem.ComputerCallOutput | ResponseFunctionWebSearch | ResponseFunctionToolCall | ResponseInputItem.FunctionCallOutput | ResponseInputItem.ToolSearchCall | ResponseToolSearchOutputItemParam | ResponseInputItem.AdditionalTools | ResponseReasoningItem | ResponseCompactionItemParam | ResponseInputItem.ImageGenerationCall | ResponseCodeInterpreterToolCall | ResponseInputItem.LocalShellCall | ResponseInputItem.LocalShellCallOutput | ResponseInputItem.ShellCall | ResponseInputItem.ShellCallOutput | ResponseInputItem.ApplyPatchCall | ResponseInputItem.ApplyPatchCallOutput | ResponseInputItem.McpListTools | ResponseInputItem.McpApprovalRequest | ResponseInputItem.McpApprovalResponse | ResponseInputItem.McpCall | ResponseCustomToolCallOutput | ResponseCustomToolCall | ResponseInputItem.CompactionTrigger | ResponseInputItem.ConfigurationUpdate | ResponseInputItem.ItemReference;
2922
2922
  export declare namespace ResponseInputItem {
2923
2923
  /**
2924
2924
  * A message input to the model with a role indicating instruction following
@@ -3504,6 +3504,20 @@ export declare namespace ResponseInputItem {
3504
3504
  */
3505
3505
  type: "compaction_trigger";
3506
3506
  }
3507
+ /**
3508
+ * Changes reasoning effort for subsequent responses without touching the
3509
+ * request-level `reasoning.effort` (GPT-6 Astra). Must not be adjacent to
3510
+ * another `configuration_update`.
3511
+ */
3512
+ interface ConfigurationUpdate {
3513
+ /**
3514
+ * The type of the item. Always `configuration_update`.
3515
+ */
3516
+ type: "configuration_update";
3517
+ reasoning: {
3518
+ effort: string;
3519
+ };
3520
+ }
3507
3521
  /**
3508
3522
  * An internal identifier for an item to reference.
3509
3523
  */
@@ -1,7 +1,8 @@
1
1
  import type { Context, Model, OpenAICompat, ProviderSessionState, ServiceTier, StreamFunction, StreamOptions, Tool, ToolChoice } from "../types.js";
2
2
  import { type OpenAIResponsesToolChoice } from "../utils/tool-choice.js";
3
+ import { type OpenAIEffortControlState } from "./openai-configuration-update.js";
3
4
  import { type OpenAIReasoningEffortFallbackState } from "./openai-reasoning-fallback.js";
4
- import type { Tool as OpenAITool, ResponseCreateParamsStreaming, ResponseInput } from "./openai-responses-wire.js";
5
+ import type { Tool as OpenAITool, ReasoningEffort, ResponseCreateParamsStreaming, ResponseInput } from "./openai-responses-wire.js";
5
6
  import { type OpenAIPromptCacheOptions, type OpenAIStrictToolsScope, type OpenAIStrictToolsState } from "./openai-shared.js";
6
7
  export interface OpenAIResponsesOptions extends StreamOptions {
7
8
  reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
@@ -63,7 +64,11 @@ interface OpenAIResponsesProviderSessionState extends ProviderSessionState, Open
63
64
  nativeHistoryReplayWarmed: boolean;
64
65
  /** Stateful `previous_response_id` chain baselines, keyed by baseUrl/model/session. */
65
66
  chains: Map<string, OpenAIResponsesChainState>;
67
+ /** `configuration_update` effort baselines, keyed by baseUrl/model/session. */
68
+ effortControls: Map<string, OpenAIEffortControlState<ResponsesStableEffort>>;
66
69
  }
70
+ /** Wire efforts a `configuration_update` can carry: every real tier, never `none`/null. */
71
+ type ResponsesStableEffort = Exclude<ReasoningEffort, "none" | null>;
67
72
  interface OpenAIResponsesChainState {
68
73
  /**
69
74
  * Wire params of the last successful turn; never carries
@@ -493,6 +493,18 @@ export interface BuildResponsesInputOptions<TApi extends Api> {
493
493
  */
494
494
  export declare function escapeReplayedControlTokens(items: ResponseInput): ResponseInput;
495
495
  export declare function buildResponsesInput<TApi extends Api>(options: BuildResponsesInputOptions<TApi>): ResponseInput;
496
+ /**
497
+ * Non-empty `reasoning_text` shipped for a synthesized reasoning item when no
498
+ * thinking text survived history reconstruction. DeepSeek-family Responses
499
+ * targets (e.g. opencode-go) reject BOTH a missing reasoning item and one whose
500
+ * `reasoning_text` is empty — "The reasoning_text in the thinking mode must be
501
+ * passed back to the API" (#8248 covered the missing case, #10690 the empty
502
+ * one). The item's presence plus a non-empty payload is what satisfies the
503
+ * contract; the exact text is immaterial once the source turn's reasoning is
504
+ * gone. Kept out of `reasoning_content="."`-territory since DeepSeek rejects the
505
+ * bare-dot synthetic placeholder on the chat-completions path.
506
+ */
507
+ export declare const SYNTHETIC_REASONING_REPLAY_PLACEHOLDER = "reasoning unavailable";
496
508
  export declare function convertResponsesAssistantMessage<TApi extends Api>(assistantMsg: AssistantMessage, model: Model<TApi>, msgIndex: number, knownCallIds: Set<string>, includeThinkingSignatures?: boolean, customCallIds?: Set<string>, preserveMessageIds?: boolean, supportsCustomToolCalls?: boolean, customToolWireNameMap?: ReadonlyMap<string, string>, computerCallIds?: Set<string>, requiresReasoningReplayForAllTurns?: boolean, requiresReasoningReplayForToolCalls?: boolean): ResponseInput;
497
509
  /**
498
510
  * Responses wire output for a tool result plus its text-only fallback.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-ai",
4
- "version": "18.1.6",
4
+ "version": "18.1.8",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Stencil Labs, Inc.",
@@ -37,10 +37,10 @@
37
37
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
38
38
  },
39
39
  "dependencies": {
40
- "@oh-my-pi/omptype": "18.1.6",
41
- "@oh-my-pi/pi-catalog": "18.1.6",
42
- "@oh-my-pi/pi-utils": "18.1.6",
43
- "@oh-my-pi/pi-wire": "18.1.6"
40
+ "@oh-my-pi/omptype": "18.1.8",
41
+ "@oh-my-pi/pi-catalog": "18.1.8",
42
+ "@oh-my-pi/pi-utils": "18.1.8",
43
+ "@oh-my-pi/pi-wire": "18.1.8"
44
44
  },
45
45
  "devDependencies": {
46
46
  "@types/bun": "^1.3.14"
@@ -5,6 +5,7 @@ import {
5
5
  applyCodexResidencyHeader,
6
6
  CODEX_BASE_URL,
7
7
  CODEX_CLIENT_VERSION,
8
+ codexRoutingHint,
8
9
  getCodexAccountId,
9
10
  OPENAI_HEADER_VALUES,
10
11
  OPENAI_HEADERS,
@@ -74,11 +75,17 @@ import {
74
75
  type CodexReasoningContext,
75
76
  type CodexRequestOptions,
76
77
  type InputItem,
78
+ type ReasoningConfig,
77
79
  type RequestBody,
78
80
  resolveCodexResponsesLite,
79
81
  transformRequestBody,
80
82
  } from "./openai-codex/request-transformer";
81
83
  import { CodexApiError } from "./openai-codex/response-handler";
84
+ import {
85
+ getOpenAIEffortControlState,
86
+ type OpenAIEffortControlState,
87
+ planStableOpenAIEffort,
88
+ } from "./openai-configuration-update";
82
89
  import type {
83
90
  ResponseComputerToolCall,
84
91
  ResponseCustomToolCall,
@@ -476,6 +483,8 @@ interface CodexProviderSessionState extends ProviderSessionState {
476
483
  webSocketSessions: Map<string, CodexWebSocketSessionState>;
477
484
  webSocketPublicToPrivate: Map<string, string>;
478
485
  metadataSessions: Map<string, CodexMetadataSessionState>;
486
+ /** `configuration_update` effort baselines, keyed by model + session. */
487
+ effortControls: Map<string, OpenAIEffortControlState<ReasoningConfig["effort"]>>;
479
488
  }
480
489
 
481
490
  /** Request classification encoded in Codex turn metadata. */
@@ -1051,6 +1060,7 @@ function createCodexProviderSessionState(): CodexProviderSessionState {
1051
1060
  webSocketSessions: new Map(),
1052
1061
  webSocketPublicToPrivate: new Map(),
1053
1062
  metadataSessions: new Map(),
1063
+ effortControls: new Map(),
1054
1064
  close: () => {
1055
1065
  for (const session of state.webSocketSessions.values()) {
1056
1066
  session.connection?.close("session_disposed");
@@ -1058,6 +1068,7 @@ function createCodexProviderSessionState(): CodexProviderSessionState {
1058
1068
  state.webSocketSessions.clear();
1059
1069
  state.webSocketPublicToPrivate.clear();
1060
1070
  state.metadataSessions.clear();
1071
+ state.effortControls.clear();
1061
1072
  },
1062
1073
  };
1063
1074
  return state;
@@ -1071,7 +1082,9 @@ function isCodexProviderSessionState(state: ProviderSessionState | undefined): s
1071
1082
  "webSocketPublicToPrivate" in state &&
1072
1083
  state.webSocketPublicToPrivate instanceof Map &&
1073
1084
  "metadataSessions" in state &&
1074
- state.metadataSessions instanceof Map
1085
+ state.metadataSessions instanceof Map &&
1086
+ "effortControls" in state &&
1087
+ state.effortControls instanceof Map
1075
1088
  );
1076
1089
  }
1077
1090
 
@@ -1573,7 +1586,30 @@ export async function buildTransformedCodexRequestBody(
1573
1586
  responsesLite: options?.responsesLite,
1574
1587
  };
1575
1588
 
1576
- return transformRequestBody(params, model, codexOptions, { developerMessages });
1589
+ const body = await transformRequestBody(params, model, codexOptions, { developerMessages });
1590
+ applyCodexStableEffort(model, body, options);
1591
+ return body;
1592
+ }
1593
+
1594
+ /**
1595
+ * Keep the request-level effort byte-stable across a conversation and carry
1596
+ * later changes as `configuration_update` items (GPT-6 Astra). Requires a
1597
+ * session id and provider session state to remember the baseline; without
1598
+ * them every request stands alone and sends its own effort.
1599
+ */
1600
+ function applyCodexStableEffort(
1601
+ model: Model<"openai-codex-responses">,
1602
+ body: RequestBody,
1603
+ options: OpenAICodexResponsesOptions | undefined,
1604
+ ): void {
1605
+ if (!model.compat.supportsConfigurationUpdate || options?.codexCompaction) return;
1606
+ const effort = body.reasoning?.effort;
1607
+ if (effort === undefined || effort === "none" || !body.input) return;
1608
+ const providerState = getCodexProviderSessionState(options?.providerSessionState);
1609
+ const sessionId = normalizeOpenAIPromptCacheKey(options?.sessionId);
1610
+ if (!providerState || !sessionId) return;
1611
+ const state = getOpenAIEffortControlState(providerState.effortControls, `${model.id}\u0000${sessionId}`);
1612
+ body.reasoning = { ...body.reasoning, effort: planStableOpenAIEffort(state, body.input, effort) };
1577
1613
  }
1578
1614
 
1579
1615
  async function openInitialCodexEventStream(
@@ -1800,6 +1836,7 @@ async function openCodexWebSocketTransport(
1800
1836
  requestContext.responsesLite,
1801
1837
  requestContext.requestMetadata,
1802
1838
  await getCodexAttestationHeader(requestContext.accountId),
1839
+ requestContext.transformedBody,
1803
1840
  );
1804
1841
  const requestBodyForState = structuredCloneJSON(requestContext.transformedBody);
1805
1842
  // `onPayload` may rewrite the outgoing frame (e.g. drop `stream_options`);
@@ -4273,6 +4310,7 @@ async function openCodexSseEventStream(
4273
4310
  responsesLite,
4274
4311
  requestMetadata,
4275
4312
  await getCodexAttestationHeader(accountId),
4313
+ body,
4276
4314
  );
4277
4315
  // `wrapCodexSseStream` arms the iterator-level idle watchdog only after this
4278
4316
  // fetch resolves. Each transport attempt needs its own pre-response timer:
@@ -4367,11 +4405,19 @@ function createCodexHeaders(
4367
4405
  responsesLite = false,
4368
4406
  requestMetadata?: CodexCompatibilityIdentity,
4369
4407
  attestation?: string,
4408
+ routedRequest?: Pick<RequestBody, "model" | "service_tier">,
4370
4409
  ): Headers {
4371
4410
  const headers = new Headers(initHeaders ?? {});
4372
4411
  headers.delete("x-api-key");
4373
4412
  headers.set("Authorization", `Bearer ${accessToken}`);
4374
4413
  if (accountId) headers.set(OPENAI_HEADERS.ACCOUNT_ID, accountId);
4414
+ // codex-rs sends the hint on every ChatGPT-OAuth request and WebSocket
4415
+ // handshake; this provider only ever speaks to the Codex backend.
4416
+ if (routedRequest) {
4417
+ headers.set(OPENAI_HEADERS.ROUTING_HINT, codexRoutingHint(routedRequest.model, routedRequest.service_tier));
4418
+ } else {
4419
+ headers.delete(OPENAI_HEADERS.ROUTING_HINT);
4420
+ }
4375
4421
  // Region-pinned enterprise workspaces answer 401 `Workspace is not authorized
4376
4422
  // in this region.` when the request's egress region does not match them and
4377
4423
  // the client did not declare the workspace's residency. The access token
@@ -0,0 +1,170 @@
1
+ /**
2
+ * Mid-conversation reasoning effort via `configuration_update` input items
3
+ * (GPT-6 Astra; `model.compat.supportsConfigurationUpdate`).
4
+ *
5
+ * The request-level `reasoning.effort` is pinned to the value of the session's
6
+ * first request so the cached prompt prefix survives an effort change. Each
7
+ * later change is carried as a `configuration_update` item inserted at the
8
+ * tail of the transcript — before the user message it takes effect on, or
9
+ * after the latest tool result when the level changes inside a tool loop — and
10
+ * replayed at that position on every subsequent request until another update
11
+ * overrides it. Mirrors the Anthropic provider's stable `output_config.effort`
12
+ * planning.
13
+ *
14
+ * Used by both the platform Responses provider and the Codex provider; the
15
+ * state lives in each provider's session state, keyed per conversation.
16
+ *
17
+ * Wire constraints (verified against the Codex backend): only `gpt-6-astra`
18
+ * accepts the item type, consecutive updates are rejected, and
19
+ * `/responses/compact` rejects histories containing them — compaction
20
+ * requests are built outside this planner and never carry the items.
21
+ */
22
+
23
+ /** `configuration_update` input item; only `reasoning.effort` is updatable. */
24
+ export interface ConfigurationUpdateItem {
25
+ type: "configuration_update";
26
+ reasoning: { effort: string };
27
+ }
28
+
29
+ interface EffortTransition<TEffort extends string> {
30
+ /** Input-array position the item is spliced into (before `input[index]`). */
31
+ index: number;
32
+ /** Fingerprint of `input[index - 1]` at record time; a mismatch means the history was rewritten. */
33
+ anchor: string;
34
+ effort: TEffort;
35
+ }
36
+
37
+ /** Per-conversation effort baseline and recorded transitions. */
38
+ export interface OpenAIEffortControlState<TEffort extends string = string> {
39
+ baseEffort?: TEffort;
40
+ currentEffort?: TEffort;
41
+ transitions: EffortTransition<TEffort>[];
42
+ }
43
+
44
+ export function createOpenAIEffortControlState<TEffort extends string>(): OpenAIEffortControlState<TEffort> {
45
+ return { transitions: [] };
46
+ }
47
+
48
+ const MAX_EFFORT_CONTROL_STATES = 16;
49
+
50
+ /**
51
+ * Fetch (or create) the control state for one conversation from a provider's
52
+ * bounded per-session map, refreshing its LRU slot.
53
+ */
54
+ export function getOpenAIEffortControlState<TEffort extends string>(
55
+ states: Map<string, OpenAIEffortControlState<TEffort>>,
56
+ key: string,
57
+ ): OpenAIEffortControlState<TEffort> {
58
+ const existing = states.get(key);
59
+ if (existing) {
60
+ states.delete(key);
61
+ states.set(key, existing);
62
+ return existing;
63
+ }
64
+ const created = createOpenAIEffortControlState<TEffort>();
65
+ states.set(key, created);
66
+ if (states.size > MAX_EFFORT_CONTROL_STATES) {
67
+ const oldest = states.keys().next().value;
68
+ if (oldest !== undefined) states.delete(oldest);
69
+ }
70
+ return created;
71
+ }
72
+
73
+ interface AnchorableItem {
74
+ type?: string | null;
75
+ role?: string;
76
+ id?: string | null;
77
+ status?: string | null;
78
+ }
79
+
80
+ /**
81
+ * Fingerprint of the item a transition sits after. Output-only lifecycle
82
+ * fields are excluded: a live response item carries `id`/`status` that the
83
+ * sanitized replay of the same item drops.
84
+ */
85
+ function effortControlAnchor(input: readonly AnchorableItem[], index: number): string {
86
+ if (index === 0) return "";
87
+ const item = input[index - 1];
88
+ if (!item) return "";
89
+ const { id: _id, status: _status, ...stable } = item;
90
+ return String(Bun.hash(JSON.stringify(stable)));
91
+ }
92
+
93
+ function resetOpenAIEffortControlState(state: OpenAIEffortControlState<string>): void {
94
+ state.baseEffort = undefined;
95
+ state.currentEffort = undefined;
96
+ state.transitions = [];
97
+ }
98
+
99
+ /**
100
+ * Discard the baseline when the request no longer continues the conversation
101
+ * it was captured for: a wire history that shrank or was rewritten under a
102
+ * recorded transition (compaction, branch switch, `/clear`). The next request
103
+ * re-baselines from its own effort, which is what the API asks for after
104
+ * compaction anyway.
105
+ */
106
+ function syncOpenAIEffortControlState(state: OpenAIEffortControlState<string>, input: readonly AnchorableItem[]): void {
107
+ for (const transition of state.transitions) {
108
+ if (transition.index > input.length || transition.anchor !== effortControlAnchor(input, transition.index)) {
109
+ resetOpenAIEffortControlState(state);
110
+ return;
111
+ }
112
+ }
113
+ }
114
+
115
+ /**
116
+ * Pin the request-level effort to the session baseline and splice pending
117
+ * `configuration_update` items into `input` (mutated in place).
118
+ *
119
+ * `input` is the freshly built transcript for this request, without any
120
+ * `configuration_update` items. `requested` is the wire effort the caller
121
+ * would otherwise send at the request level. Returns the effort to send at the
122
+ * request level (`requested` on the first request, the baseline afterwards).
123
+ */
124
+ export function planStableOpenAIEffort<TItem extends AnchorableItem, TEffort extends string>(
125
+ state: OpenAIEffortControlState<TEffort>,
126
+ input: Array<TItem | ConfigurationUpdateItem>,
127
+ requested: TEffort,
128
+ ): TEffort {
129
+ syncOpenAIEffortControlState(state, input);
130
+ if (state.baseEffort === undefined) {
131
+ state.baseEffort = requested;
132
+ state.currentEffort = requested;
133
+ return requested;
134
+ }
135
+ if (state.currentEffort !== requested) {
136
+ const last = input[input.length - 1];
137
+ const index = last && "role" in last && last.role === "user" ? input.length - 1 : input.length;
138
+ const existing = state.transitions.find(transition => transition.index === index);
139
+ if (existing) {
140
+ existing.effort = requested;
141
+ } else {
142
+ state.transitions.push({ index, anchor: effortControlAnchor(input, index), effort: requested });
143
+ }
144
+ // A change back to the effort already in force at that position is a
145
+ // no-op on the wire; drop it rather than send a redundant item.
146
+ let preceding = state.baseEffort;
147
+ let precedingIndex = -1;
148
+ for (const transition of state.transitions) {
149
+ if (transition.index < index && transition.index > precedingIndex) {
150
+ preceding = transition.effort;
151
+ precedingIndex = transition.index;
152
+ }
153
+ }
154
+ if (requested === preceding) {
155
+ state.transitions = state.transitions.filter(transition => transition.index !== index);
156
+ }
157
+ state.currentEffort = requested;
158
+ }
159
+ // Splice in ascending order so each insertion offsets only the ones after it.
160
+ state.transitions.sort((a, b) => a.index - b.index);
161
+ let offset = 0;
162
+ for (const transition of state.transitions) {
163
+ input.splice(transition.index + offset, 0, {
164
+ type: "configuration_update",
165
+ reasoning: { effort: transition.effort },
166
+ });
167
+ offset++;
168
+ }
169
+ return state.baseEffort;
170
+ }
@@ -3013,6 +3013,7 @@ export type ResponseInputItem =
3013
3013
  | ResponseCustomToolCallOutput
3014
3014
  | ResponseCustomToolCall
3015
3015
  | ResponseInputItem.CompactionTrigger
3016
+ | ResponseInputItem.ConfigurationUpdate
3016
3017
  | ResponseInputItem.ItemReference;
3017
3018
  export declare namespace ResponseInputItem {
3018
3019
  /**
@@ -3599,6 +3600,20 @@ export declare namespace ResponseInputItem {
3599
3600
  */
3600
3601
  type: "compaction_trigger";
3601
3602
  }
3603
+ /**
3604
+ * Changes reasoning effort for subsequent responses without touching the
3605
+ * request-level `reasoning.effort` (GPT-6 Astra). Must not be adjacent to
3606
+ * another `configuration_update`.
3607
+ */
3608
+ interface ConfigurationUpdate {
3609
+ /**
3610
+ * The type of the item. Always `configuration_update`.
3611
+ */
3612
+ type: "configuration_update";
3613
+ reasoning: {
3614
+ effort: string;
3615
+ };
3616
+ }
3602
3617
  /**
3603
3618
  * An internal identifier for an item to reference.
3604
3619
  */
@@ -48,6 +48,11 @@ import {
48
48
  type OpenAIResponsesToolChoice,
49
49
  } from "../utils/tool-choice";
50
50
  import { compactGrammarDefinition } from "./grammar";
51
+ import {
52
+ getOpenAIEffortControlState,
53
+ type OpenAIEffortControlState,
54
+ planStableOpenAIEffort,
55
+ } from "./openai-configuration-update";
51
56
  import {
52
57
  applyOpenAIReasoningEffortFallback,
53
58
  clearOpenAIReasoningEffortFallbackState,
@@ -61,6 +66,7 @@ import {
61
66
  } from "./openai-reasoning-fallback";
62
67
  import type {
63
68
  Tool as OpenAITool,
69
+ ReasoningEffort,
64
70
  ResponseCreateParamsStreaming,
65
71
  ResponseInput,
66
72
  ResponseInputContent,
@@ -197,8 +203,13 @@ interface OpenAIResponsesProviderSessionState
197
203
  nativeHistoryReplayWarmed: boolean;
198
204
  /** Stateful `previous_response_id` chain baselines, keyed by baseUrl/model/session. */
199
205
  chains: Map<string, OpenAIResponsesChainState>;
206
+ /** `configuration_update` effort baselines, keyed by baseUrl/model/session. */
207
+ effortControls: Map<string, OpenAIEffortControlState<ResponsesStableEffort>>;
200
208
  }
201
209
 
210
+ /** Wire efforts a `configuration_update` can carry: every real tier, never `none`/null. */
211
+ type ResponsesStableEffort = Exclude<ReasoningEffort, "none" | null>;
212
+
202
213
  interface OpenAIResponsesChainState {
203
214
  /**
204
215
  * Wire params of the last successful turn; never carries
@@ -224,9 +235,11 @@ function createOpenAIResponsesProviderSessionState(): OpenAIResponsesProviderSes
224
235
  ...reasoningEffortFallbackState,
225
236
  nativeHistoryReplayWarmed: false,
226
237
  chains: new Map(),
238
+ effortControls: new Map(),
227
239
  close: () => {
228
240
  state.nativeHistoryReplayWarmed = false;
229
241
  state.chains.clear();
242
+ state.effortControls.clear();
230
243
  clearOpenAIStrictToolsState(state);
231
244
  clearOpenAIReasoningEffortFallbackState(state);
232
245
  },
@@ -1275,6 +1288,7 @@ export function buildParams(
1275
1288
  if (model.reasoningMode && !options?.forceReasoningOff) {
1276
1289
  params.reasoning = { ...params.reasoning, mode: model.reasoningMode };
1277
1290
  }
1291
+ applyResponsesStableEffort(model, params, messages, options, providerSessionState);
1278
1292
 
1279
1293
  if (model.compat.isVercelGatewayHost) {
1280
1294
  applyVercelResponsesCacheControls(params, model.compat, cacheRetention);
@@ -1299,6 +1313,33 @@ export function buildParams(
1299
1313
  return { params, trailingScaffoldingItems, strictToolsApplied };
1300
1314
  }
1301
1315
 
1316
+ /**
1317
+ * Keep the request-level effort byte-stable across a conversation and carry
1318
+ * later changes as `configuration_update` items (GPT-6 Astra). Requires a
1319
+ * routing session id and provider session state to remember the baseline;
1320
+ * without them every request stands alone and sends its own effort.
1321
+ */
1322
+ function applyResponsesStableEffort(
1323
+ model: Model<"openai-responses">,
1324
+ params: OpenAIResponsesSamplingParams,
1325
+ input: ResponseInput,
1326
+ options: OpenAIResponsesOptions | undefined,
1327
+ providerSessionState: OpenAIResponsesProviderSessionState | undefined,
1328
+ ): void {
1329
+ if (!model.compat.supportsConfigurationUpdate || !providerSessionState) return;
1330
+ const reasoning = params.reasoning;
1331
+ if (!reasoning || !("effort" in reasoning)) return;
1332
+ const effort = reasoning.effort;
1333
+ if (effort === undefined || effort === null || effort === "none") return;
1334
+ const sessionId = getOpenAIResponsesRoutingSessionId(options);
1335
+ if (!sessionId) return;
1336
+ const state = getOpenAIEffortControlState(
1337
+ providerSessionState.effortControls,
1338
+ `${model.baseUrl ?? ""}\u0000${model.id}\u0000${sessionId}`,
1339
+ );
1340
+ params.reasoning = { ...reasoning, effort: planStableOpenAIEffort(state, input, effort) };
1341
+ }
1342
+
1302
1343
  /**
1303
1344
  * Whether this model should get the OpenAI custom-tool grammar variant
1304
1345
  * for `apply_patch`. The generated model catalog sets
@@ -2111,6 +2111,19 @@ function parseResponseReasoningReplayItem(signature: string | undefined): Respon
2111
2111
  }
2112
2112
  }
2113
2113
 
2114
+ /**
2115
+ * Non-empty `reasoning_text` shipped for a synthesized reasoning item when no
2116
+ * thinking text survived history reconstruction. DeepSeek-family Responses
2117
+ * targets (e.g. opencode-go) reject BOTH a missing reasoning item and one whose
2118
+ * `reasoning_text` is empty — "The reasoning_text in the thinking mode must be
2119
+ * passed back to the API" (#8248 covered the missing case, #10690 the empty
2120
+ * one). The item's presence plus a non-empty payload is what satisfies the
2121
+ * contract; the exact text is immaterial once the source turn's reasoning is
2122
+ * gone. Kept out of `reasoning_content="."`-territory since DeepSeek rejects the
2123
+ * bare-dot synthetic placeholder on the chat-completions path.
2124
+ */
2125
+ export const SYNTHETIC_REASONING_REPLAY_PLACEHOLDER = "reasoning unavailable";
2126
+
2114
2127
  export function convertResponsesAssistantMessage<TApi extends Api>(
2115
2128
  assistantMsg: AssistantMessage,
2116
2129
  model: Model<TApi>,
@@ -2268,12 +2281,15 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
2268
2281
  if (requiresReasoningItem && !reasoningItemEmitted && outputItems.length > 0) {
2269
2282
  // Replay the demoted reasoning (already present in `content` as visible
2270
2283
  // text) as a structured reasoning item so the thinking-mode continuation
2271
- // carries the `reasoning_text` the provider requires. The text may be empty
2272
- // when the source turn was minted by another model and its reasoning is
2273
- // already folded into the message text; the item's presence is what
2274
- // satisfies the provider contract, mirroring the empty `reasoning_content`
2275
- // placeholder used on the chat-completions path.
2276
- const reasoningText = carriedReasoningTexts.join("\n");
2284
+ // carries the `reasoning_text` the provider requires. When no thinking
2285
+ // text survived reconstruction (source turn minted by another model, or
2286
+ // reasoning dropped by compaction/archive budget) the carried text is
2287
+ // empty — and DeepSeek-family targets reject an empty `reasoning_text`
2288
+ // exactly like a missing item (#10690), so substitute a non-empty
2289
+ // placeholder. The `id` still prefers a surviving upstream item id.
2290
+ const carriedReasoningText = carriedReasoningTexts.join("\n");
2291
+ const reasoningText =
2292
+ carriedReasoningText.length > 0 ? carriedReasoningText : SYNTHETIC_REASONING_REPLAY_PLACEHOLDER;
2277
2293
  const reasoningId =
2278
2294
  synthesizedReasoningItemId ?? `rs_${Bun.hash(`${model.id}:${msgIndex}:${reasoningText}`).toString(36)}`;
2279
2295
  const reasoningItem: ResponseReasoningItem = {