@oh-my-pi/pi-ai 18.1.6 → 18.1.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +16 -0
- package/dist/types/providers/openai-configuration-update.d.ts +64 -0
- package/dist/types/providers/openai-responses-wire.d.ts +15 -1
- package/dist/types/providers/openai-responses.d.ts +6 -1
- package/dist/types/providers/openai-shared.d.ts +12 -0
- package/package.json +5 -5
- package/src/providers/openai-codex-responses.ts +48 -2
- package/src/providers/openai-configuration-update.ts +170 -0
- package/src/providers/openai-responses-wire.ts +15 -0
- package/src/providers/openai-responses.ts +41 -0
- package/src/providers/openai-shared.ts +22 -6
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,22 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.1.8] - 2026-09-03
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- Added GPT-6 Astra support for preserving prompt caching when changing the thinking level during a conversation across the OpenAI and OpenAI Codex providers.
|
|
10
|
+
|
|
11
|
+
### Changed
|
|
12
|
+
|
|
13
|
+
- Updated OpenAI Codex requests to improve routing by communicating the selected model and service tier across Responses, WebSocket, and remote-compaction requests.
|
|
14
|
+
|
|
15
|
+
## [18.1.7] - 2026-09-03
|
|
16
|
+
|
|
17
|
+
### Fixed
|
|
18
|
+
|
|
19
|
+
- Fixed DeepSeek-family Responses replay (e.g. opencode-go) rejecting a resumed thinking-mode turn with `400 The reasoning_text in the thinking mode must be passed back to the API` when compaction dropped the turn's reasoning; a non-empty placeholder is now synthesized instead of an empty `reasoning_text` ([#10690](https://github.com/can1357/oh-my-pi/issues/10690)).
|
|
20
|
+
|
|
5
21
|
## [18.1.6] - 2026-09-03
|
|
6
22
|
|
|
7
23
|
### Breaking Changes
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Mid-conversation reasoning effort via `configuration_update` input items
|
|
3
|
+
* (GPT-6 Astra; `model.compat.supportsConfigurationUpdate`).
|
|
4
|
+
*
|
|
5
|
+
* The request-level `reasoning.effort` is pinned to the value of the session's
|
|
6
|
+
* first request so the cached prompt prefix survives an effort change. Each
|
|
7
|
+
* later change is carried as a `configuration_update` item inserted at the
|
|
8
|
+
* tail of the transcript — before the user message it takes effect on, or
|
|
9
|
+
* after the latest tool result when the level changes inside a tool loop — and
|
|
10
|
+
* replayed at that position on every subsequent request until another update
|
|
11
|
+
* overrides it. Mirrors the Anthropic provider's stable `output_config.effort`
|
|
12
|
+
* planning.
|
|
13
|
+
*
|
|
14
|
+
* Used by both the platform Responses provider and the Codex provider; the
|
|
15
|
+
* state lives in each provider's session state, keyed per conversation.
|
|
16
|
+
*
|
|
17
|
+
* Wire constraints (verified against the Codex backend): only `gpt-6-astra`
|
|
18
|
+
* accepts the item type, consecutive updates are rejected, and
|
|
19
|
+
* `/responses/compact` rejects histories containing them — compaction
|
|
20
|
+
* requests are built outside this planner and never carry the items.
|
|
21
|
+
*/
|
|
22
|
+
/** `configuration_update` input item; only `reasoning.effort` is updatable. */
|
|
23
|
+
export interface ConfigurationUpdateItem {
|
|
24
|
+
type: "configuration_update";
|
|
25
|
+
reasoning: {
|
|
26
|
+
effort: string;
|
|
27
|
+
};
|
|
28
|
+
}
|
|
29
|
+
interface EffortTransition<TEffort extends string> {
|
|
30
|
+
/** Input-array position the item is spliced into (before `input[index]`). */
|
|
31
|
+
index: number;
|
|
32
|
+
/** Fingerprint of `input[index - 1]` at record time; a mismatch means the history was rewritten. */
|
|
33
|
+
anchor: string;
|
|
34
|
+
effort: TEffort;
|
|
35
|
+
}
|
|
36
|
+
/** Per-conversation effort baseline and recorded transitions. */
|
|
37
|
+
export interface OpenAIEffortControlState<TEffort extends string = string> {
|
|
38
|
+
baseEffort?: TEffort;
|
|
39
|
+
currentEffort?: TEffort;
|
|
40
|
+
transitions: EffortTransition<TEffort>[];
|
|
41
|
+
}
|
|
42
|
+
export declare function createOpenAIEffortControlState<TEffort extends string>(): OpenAIEffortControlState<TEffort>;
|
|
43
|
+
/**
|
|
44
|
+
* Fetch (or create) the control state for one conversation from a provider's
|
|
45
|
+
* bounded per-session map, refreshing its LRU slot.
|
|
46
|
+
*/
|
|
47
|
+
export declare function getOpenAIEffortControlState<TEffort extends string>(states: Map<string, OpenAIEffortControlState<TEffort>>, key: string): OpenAIEffortControlState<TEffort>;
|
|
48
|
+
interface AnchorableItem {
|
|
49
|
+
type?: string | null;
|
|
50
|
+
role?: string;
|
|
51
|
+
id?: string | null;
|
|
52
|
+
status?: string | null;
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* Pin the request-level effort to the session baseline and splice pending
|
|
56
|
+
* `configuration_update` items into `input` (mutated in place).
|
|
57
|
+
*
|
|
58
|
+
* `input` is the freshly built transcript for this request, without any
|
|
59
|
+
* `configuration_update` items. `requested` is the wire effort the caller
|
|
60
|
+
* would otherwise send at the request level. Returns the effort to send at the
|
|
61
|
+
* request level (`requested` on the first request, the baseline afterwards).
|
|
62
|
+
*/
|
|
63
|
+
export declare function planStableOpenAIEffort<TItem extends AnchorableItem, TEffort extends string>(state: OpenAIEffortControlState<TEffort>, input: Array<TItem | ConfigurationUpdateItem>, requested: TEffort): TEffort;
|
|
64
|
+
export {};
|
|
@@ -2918,7 +2918,7 @@ export interface ResponseInputImageContent {
|
|
|
2918
2918
|
* `assistant` role are presumed to have been generated by the model in previous
|
|
2919
2919
|
* interactions.
|
|
2920
2920
|
*/
|
|
2921
|
-
export type ResponseInputItem = EasyInputMessage | ResponseInputItem.Message | ResponseOutputMessage | ResponseFileSearchToolCall | ResponseComputerToolCall | ResponseInputItem.ComputerCallOutput | ResponseFunctionWebSearch | ResponseFunctionToolCall | ResponseInputItem.FunctionCallOutput | ResponseInputItem.ToolSearchCall | ResponseToolSearchOutputItemParam | ResponseInputItem.AdditionalTools | ResponseReasoningItem | ResponseCompactionItemParam | ResponseInputItem.ImageGenerationCall | ResponseCodeInterpreterToolCall | ResponseInputItem.LocalShellCall | ResponseInputItem.LocalShellCallOutput | ResponseInputItem.ShellCall | ResponseInputItem.ShellCallOutput | ResponseInputItem.ApplyPatchCall | ResponseInputItem.ApplyPatchCallOutput | ResponseInputItem.McpListTools | ResponseInputItem.McpApprovalRequest | ResponseInputItem.McpApprovalResponse | ResponseInputItem.McpCall | ResponseCustomToolCallOutput | ResponseCustomToolCall | ResponseInputItem.CompactionTrigger | ResponseInputItem.ItemReference;
|
|
2921
|
+
export type ResponseInputItem = EasyInputMessage | ResponseInputItem.Message | ResponseOutputMessage | ResponseFileSearchToolCall | ResponseComputerToolCall | ResponseInputItem.ComputerCallOutput | ResponseFunctionWebSearch | ResponseFunctionToolCall | ResponseInputItem.FunctionCallOutput | ResponseInputItem.ToolSearchCall | ResponseToolSearchOutputItemParam | ResponseInputItem.AdditionalTools | ResponseReasoningItem | ResponseCompactionItemParam | ResponseInputItem.ImageGenerationCall | ResponseCodeInterpreterToolCall | ResponseInputItem.LocalShellCall | ResponseInputItem.LocalShellCallOutput | ResponseInputItem.ShellCall | ResponseInputItem.ShellCallOutput | ResponseInputItem.ApplyPatchCall | ResponseInputItem.ApplyPatchCallOutput | ResponseInputItem.McpListTools | ResponseInputItem.McpApprovalRequest | ResponseInputItem.McpApprovalResponse | ResponseInputItem.McpCall | ResponseCustomToolCallOutput | ResponseCustomToolCall | ResponseInputItem.CompactionTrigger | ResponseInputItem.ConfigurationUpdate | ResponseInputItem.ItemReference;
|
|
2922
2922
|
export declare namespace ResponseInputItem {
|
|
2923
2923
|
/**
|
|
2924
2924
|
* A message input to the model with a role indicating instruction following
|
|
@@ -3504,6 +3504,20 @@ export declare namespace ResponseInputItem {
|
|
|
3504
3504
|
*/
|
|
3505
3505
|
type: "compaction_trigger";
|
|
3506
3506
|
}
|
|
3507
|
+
/**
|
|
3508
|
+
* Changes reasoning effort for subsequent responses without touching the
|
|
3509
|
+
* request-level `reasoning.effort` (GPT-6 Astra). Must not be adjacent to
|
|
3510
|
+
* another `configuration_update`.
|
|
3511
|
+
*/
|
|
3512
|
+
interface ConfigurationUpdate {
|
|
3513
|
+
/**
|
|
3514
|
+
* The type of the item. Always `configuration_update`.
|
|
3515
|
+
*/
|
|
3516
|
+
type: "configuration_update";
|
|
3517
|
+
reasoning: {
|
|
3518
|
+
effort: string;
|
|
3519
|
+
};
|
|
3520
|
+
}
|
|
3507
3521
|
/**
|
|
3508
3522
|
* An internal identifier for an item to reference.
|
|
3509
3523
|
*/
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import type { Context, Model, OpenAICompat, ProviderSessionState, ServiceTier, StreamFunction, StreamOptions, Tool, ToolChoice } from "../types.js";
|
|
2
2
|
import { type OpenAIResponsesToolChoice } from "../utils/tool-choice.js";
|
|
3
|
+
import { type OpenAIEffortControlState } from "./openai-configuration-update.js";
|
|
3
4
|
import { type OpenAIReasoningEffortFallbackState } from "./openai-reasoning-fallback.js";
|
|
4
|
-
import type { Tool as OpenAITool, ResponseCreateParamsStreaming, ResponseInput } from "./openai-responses-wire.js";
|
|
5
|
+
import type { Tool as OpenAITool, ReasoningEffort, ResponseCreateParamsStreaming, ResponseInput } from "./openai-responses-wire.js";
|
|
5
6
|
import { type OpenAIPromptCacheOptions, type OpenAIStrictToolsScope, type OpenAIStrictToolsState } from "./openai-shared.js";
|
|
6
7
|
export interface OpenAIResponsesOptions extends StreamOptions {
|
|
7
8
|
reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
|
|
@@ -63,7 +64,11 @@ interface OpenAIResponsesProviderSessionState extends ProviderSessionState, Open
|
|
|
63
64
|
nativeHistoryReplayWarmed: boolean;
|
|
64
65
|
/** Stateful `previous_response_id` chain baselines, keyed by baseUrl/model/session. */
|
|
65
66
|
chains: Map<string, OpenAIResponsesChainState>;
|
|
67
|
+
/** `configuration_update` effort baselines, keyed by baseUrl/model/session. */
|
|
68
|
+
effortControls: Map<string, OpenAIEffortControlState<ResponsesStableEffort>>;
|
|
66
69
|
}
|
|
70
|
+
/** Wire efforts a `configuration_update` can carry: every real tier, never `none`/null. */
|
|
71
|
+
type ResponsesStableEffort = Exclude<ReasoningEffort, "none" | null>;
|
|
67
72
|
interface OpenAIResponsesChainState {
|
|
68
73
|
/**
|
|
69
74
|
* Wire params of the last successful turn; never carries
|
|
@@ -493,6 +493,18 @@ export interface BuildResponsesInputOptions<TApi extends Api> {
|
|
|
493
493
|
*/
|
|
494
494
|
export declare function escapeReplayedControlTokens(items: ResponseInput): ResponseInput;
|
|
495
495
|
export declare function buildResponsesInput<TApi extends Api>(options: BuildResponsesInputOptions<TApi>): ResponseInput;
|
|
496
|
+
/**
|
|
497
|
+
* Non-empty `reasoning_text` shipped for a synthesized reasoning item when no
|
|
498
|
+
* thinking text survived history reconstruction. DeepSeek-family Responses
|
|
499
|
+
* targets (e.g. opencode-go) reject BOTH a missing reasoning item and one whose
|
|
500
|
+
* `reasoning_text` is empty — "The reasoning_text in the thinking mode must be
|
|
501
|
+
* passed back to the API" (#8248 covered the missing case, #10690 the empty
|
|
502
|
+
* one). The item's presence plus a non-empty payload is what satisfies the
|
|
503
|
+
* contract; the exact text is immaterial once the source turn's reasoning is
|
|
504
|
+
* gone. Kept out of `reasoning_content="."`-territory since DeepSeek rejects the
|
|
505
|
+
* bare-dot synthetic placeholder on the chat-completions path.
|
|
506
|
+
*/
|
|
507
|
+
export declare const SYNTHETIC_REASONING_REPLAY_PLACEHOLDER = "reasoning unavailable";
|
|
496
508
|
export declare function convertResponsesAssistantMessage<TApi extends Api>(assistantMsg: AssistantMessage, model: Model<TApi>, msgIndex: number, knownCallIds: Set<string>, includeThinkingSignatures?: boolean, customCallIds?: Set<string>, preserveMessageIds?: boolean, supportsCustomToolCalls?: boolean, customToolWireNameMap?: ReadonlyMap<string, string>, computerCallIds?: Set<string>, requiresReasoningReplayForAllTurns?: boolean, requiresReasoningReplayForToolCalls?: boolean): ResponseInput;
|
|
497
509
|
/**
|
|
498
510
|
* Responses wire output for a tool result plus its text-only fallback.
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-ai",
|
|
4
|
-
"version": "18.1.
|
|
4
|
+
"version": "18.1.8",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": "Stencil Labs, Inc.",
|
|
@@ -37,10 +37,10 @@
|
|
|
37
37
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
38
38
|
},
|
|
39
39
|
"dependencies": {
|
|
40
|
-
"@oh-my-pi/omptype": "18.1.
|
|
41
|
-
"@oh-my-pi/pi-catalog": "18.1.
|
|
42
|
-
"@oh-my-pi/pi-utils": "18.1.
|
|
43
|
-
"@oh-my-pi/pi-wire": "18.1.
|
|
40
|
+
"@oh-my-pi/omptype": "18.1.8",
|
|
41
|
+
"@oh-my-pi/pi-catalog": "18.1.8",
|
|
42
|
+
"@oh-my-pi/pi-utils": "18.1.8",
|
|
43
|
+
"@oh-my-pi/pi-wire": "18.1.8"
|
|
44
44
|
},
|
|
45
45
|
"devDependencies": {
|
|
46
46
|
"@types/bun": "^1.3.14"
|
|
@@ -5,6 +5,7 @@ import {
|
|
|
5
5
|
applyCodexResidencyHeader,
|
|
6
6
|
CODEX_BASE_URL,
|
|
7
7
|
CODEX_CLIENT_VERSION,
|
|
8
|
+
codexRoutingHint,
|
|
8
9
|
getCodexAccountId,
|
|
9
10
|
OPENAI_HEADER_VALUES,
|
|
10
11
|
OPENAI_HEADERS,
|
|
@@ -74,11 +75,17 @@ import {
|
|
|
74
75
|
type CodexReasoningContext,
|
|
75
76
|
type CodexRequestOptions,
|
|
76
77
|
type InputItem,
|
|
78
|
+
type ReasoningConfig,
|
|
77
79
|
type RequestBody,
|
|
78
80
|
resolveCodexResponsesLite,
|
|
79
81
|
transformRequestBody,
|
|
80
82
|
} from "./openai-codex/request-transformer";
|
|
81
83
|
import { CodexApiError } from "./openai-codex/response-handler";
|
|
84
|
+
import {
|
|
85
|
+
getOpenAIEffortControlState,
|
|
86
|
+
type OpenAIEffortControlState,
|
|
87
|
+
planStableOpenAIEffort,
|
|
88
|
+
} from "./openai-configuration-update";
|
|
82
89
|
import type {
|
|
83
90
|
ResponseComputerToolCall,
|
|
84
91
|
ResponseCustomToolCall,
|
|
@@ -476,6 +483,8 @@ interface CodexProviderSessionState extends ProviderSessionState {
|
|
|
476
483
|
webSocketSessions: Map<string, CodexWebSocketSessionState>;
|
|
477
484
|
webSocketPublicToPrivate: Map<string, string>;
|
|
478
485
|
metadataSessions: Map<string, CodexMetadataSessionState>;
|
|
486
|
+
/** `configuration_update` effort baselines, keyed by model + session. */
|
|
487
|
+
effortControls: Map<string, OpenAIEffortControlState<ReasoningConfig["effort"]>>;
|
|
479
488
|
}
|
|
480
489
|
|
|
481
490
|
/** Request classification encoded in Codex turn metadata. */
|
|
@@ -1051,6 +1060,7 @@ function createCodexProviderSessionState(): CodexProviderSessionState {
|
|
|
1051
1060
|
webSocketSessions: new Map(),
|
|
1052
1061
|
webSocketPublicToPrivate: new Map(),
|
|
1053
1062
|
metadataSessions: new Map(),
|
|
1063
|
+
effortControls: new Map(),
|
|
1054
1064
|
close: () => {
|
|
1055
1065
|
for (const session of state.webSocketSessions.values()) {
|
|
1056
1066
|
session.connection?.close("session_disposed");
|
|
@@ -1058,6 +1068,7 @@ function createCodexProviderSessionState(): CodexProviderSessionState {
|
|
|
1058
1068
|
state.webSocketSessions.clear();
|
|
1059
1069
|
state.webSocketPublicToPrivate.clear();
|
|
1060
1070
|
state.metadataSessions.clear();
|
|
1071
|
+
state.effortControls.clear();
|
|
1061
1072
|
},
|
|
1062
1073
|
};
|
|
1063
1074
|
return state;
|
|
@@ -1071,7 +1082,9 @@ function isCodexProviderSessionState(state: ProviderSessionState | undefined): s
|
|
|
1071
1082
|
"webSocketPublicToPrivate" in state &&
|
|
1072
1083
|
state.webSocketPublicToPrivate instanceof Map &&
|
|
1073
1084
|
"metadataSessions" in state &&
|
|
1074
|
-
state.metadataSessions instanceof Map
|
|
1085
|
+
state.metadataSessions instanceof Map &&
|
|
1086
|
+
"effortControls" in state &&
|
|
1087
|
+
state.effortControls instanceof Map
|
|
1075
1088
|
);
|
|
1076
1089
|
}
|
|
1077
1090
|
|
|
@@ -1573,7 +1586,30 @@ export async function buildTransformedCodexRequestBody(
|
|
|
1573
1586
|
responsesLite: options?.responsesLite,
|
|
1574
1587
|
};
|
|
1575
1588
|
|
|
1576
|
-
|
|
1589
|
+
const body = await transformRequestBody(params, model, codexOptions, { developerMessages });
|
|
1590
|
+
applyCodexStableEffort(model, body, options);
|
|
1591
|
+
return body;
|
|
1592
|
+
}
|
|
1593
|
+
|
|
1594
|
+
/**
|
|
1595
|
+
* Keep the request-level effort byte-stable across a conversation and carry
|
|
1596
|
+
* later changes as `configuration_update` items (GPT-6 Astra). Requires a
|
|
1597
|
+
* session id and provider session state to remember the baseline; without
|
|
1598
|
+
* them every request stands alone and sends its own effort.
|
|
1599
|
+
*/
|
|
1600
|
+
function applyCodexStableEffort(
|
|
1601
|
+
model: Model<"openai-codex-responses">,
|
|
1602
|
+
body: RequestBody,
|
|
1603
|
+
options: OpenAICodexResponsesOptions | undefined,
|
|
1604
|
+
): void {
|
|
1605
|
+
if (!model.compat.supportsConfigurationUpdate || options?.codexCompaction) return;
|
|
1606
|
+
const effort = body.reasoning?.effort;
|
|
1607
|
+
if (effort === undefined || effort === "none" || !body.input) return;
|
|
1608
|
+
const providerState = getCodexProviderSessionState(options?.providerSessionState);
|
|
1609
|
+
const sessionId = normalizeOpenAIPromptCacheKey(options?.sessionId);
|
|
1610
|
+
if (!providerState || !sessionId) return;
|
|
1611
|
+
const state = getOpenAIEffortControlState(providerState.effortControls, `${model.id}\u0000${sessionId}`);
|
|
1612
|
+
body.reasoning = { ...body.reasoning, effort: planStableOpenAIEffort(state, body.input, effort) };
|
|
1577
1613
|
}
|
|
1578
1614
|
|
|
1579
1615
|
async function openInitialCodexEventStream(
|
|
@@ -1800,6 +1836,7 @@ async function openCodexWebSocketTransport(
|
|
|
1800
1836
|
requestContext.responsesLite,
|
|
1801
1837
|
requestContext.requestMetadata,
|
|
1802
1838
|
await getCodexAttestationHeader(requestContext.accountId),
|
|
1839
|
+
requestContext.transformedBody,
|
|
1803
1840
|
);
|
|
1804
1841
|
const requestBodyForState = structuredCloneJSON(requestContext.transformedBody);
|
|
1805
1842
|
// `onPayload` may rewrite the outgoing frame (e.g. drop `stream_options`);
|
|
@@ -4273,6 +4310,7 @@ async function openCodexSseEventStream(
|
|
|
4273
4310
|
responsesLite,
|
|
4274
4311
|
requestMetadata,
|
|
4275
4312
|
await getCodexAttestationHeader(accountId),
|
|
4313
|
+
body,
|
|
4276
4314
|
);
|
|
4277
4315
|
// `wrapCodexSseStream` arms the iterator-level idle watchdog only after this
|
|
4278
4316
|
// fetch resolves. Each transport attempt needs its own pre-response timer:
|
|
@@ -4367,11 +4405,19 @@ function createCodexHeaders(
|
|
|
4367
4405
|
responsesLite = false,
|
|
4368
4406
|
requestMetadata?: CodexCompatibilityIdentity,
|
|
4369
4407
|
attestation?: string,
|
|
4408
|
+
routedRequest?: Pick<RequestBody, "model" | "service_tier">,
|
|
4370
4409
|
): Headers {
|
|
4371
4410
|
const headers = new Headers(initHeaders ?? {});
|
|
4372
4411
|
headers.delete("x-api-key");
|
|
4373
4412
|
headers.set("Authorization", `Bearer ${accessToken}`);
|
|
4374
4413
|
if (accountId) headers.set(OPENAI_HEADERS.ACCOUNT_ID, accountId);
|
|
4414
|
+
// codex-rs sends the hint on every ChatGPT-OAuth request and WebSocket
|
|
4415
|
+
// handshake; this provider only ever speaks to the Codex backend.
|
|
4416
|
+
if (routedRequest) {
|
|
4417
|
+
headers.set(OPENAI_HEADERS.ROUTING_HINT, codexRoutingHint(routedRequest.model, routedRequest.service_tier));
|
|
4418
|
+
} else {
|
|
4419
|
+
headers.delete(OPENAI_HEADERS.ROUTING_HINT);
|
|
4420
|
+
}
|
|
4375
4421
|
// Region-pinned enterprise workspaces answer 401 `Workspace is not authorized
|
|
4376
4422
|
// in this region.` when the request's egress region does not match them and
|
|
4377
4423
|
// the client did not declare the workspace's residency. The access token
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Mid-conversation reasoning effort via `configuration_update` input items
|
|
3
|
+
* (GPT-6 Astra; `model.compat.supportsConfigurationUpdate`).
|
|
4
|
+
*
|
|
5
|
+
* The request-level `reasoning.effort` is pinned to the value of the session's
|
|
6
|
+
* first request so the cached prompt prefix survives an effort change. Each
|
|
7
|
+
* later change is carried as a `configuration_update` item inserted at the
|
|
8
|
+
* tail of the transcript — before the user message it takes effect on, or
|
|
9
|
+
* after the latest tool result when the level changes inside a tool loop — and
|
|
10
|
+
* replayed at that position on every subsequent request until another update
|
|
11
|
+
* overrides it. Mirrors the Anthropic provider's stable `output_config.effort`
|
|
12
|
+
* planning.
|
|
13
|
+
*
|
|
14
|
+
* Used by both the platform Responses provider and the Codex provider; the
|
|
15
|
+
* state lives in each provider's session state, keyed per conversation.
|
|
16
|
+
*
|
|
17
|
+
* Wire constraints (verified against the Codex backend): only `gpt-6-astra`
|
|
18
|
+
* accepts the item type, consecutive updates are rejected, and
|
|
19
|
+
* `/responses/compact` rejects histories containing them — compaction
|
|
20
|
+
* requests are built outside this planner and never carry the items.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
/** `configuration_update` input item; only `reasoning.effort` is updatable. */
|
|
24
|
+
export interface ConfigurationUpdateItem {
|
|
25
|
+
type: "configuration_update";
|
|
26
|
+
reasoning: { effort: string };
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
interface EffortTransition<TEffort extends string> {
|
|
30
|
+
/** Input-array position the item is spliced into (before `input[index]`). */
|
|
31
|
+
index: number;
|
|
32
|
+
/** Fingerprint of `input[index - 1]` at record time; a mismatch means the history was rewritten. */
|
|
33
|
+
anchor: string;
|
|
34
|
+
effort: TEffort;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** Per-conversation effort baseline and recorded transitions. */
|
|
38
|
+
export interface OpenAIEffortControlState<TEffort extends string = string> {
|
|
39
|
+
baseEffort?: TEffort;
|
|
40
|
+
currentEffort?: TEffort;
|
|
41
|
+
transitions: EffortTransition<TEffort>[];
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export function createOpenAIEffortControlState<TEffort extends string>(): OpenAIEffortControlState<TEffort> {
|
|
45
|
+
return { transitions: [] };
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const MAX_EFFORT_CONTROL_STATES = 16;
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Fetch (or create) the control state for one conversation from a provider's
|
|
52
|
+
* bounded per-session map, refreshing its LRU slot.
|
|
53
|
+
*/
|
|
54
|
+
export function getOpenAIEffortControlState<TEffort extends string>(
|
|
55
|
+
states: Map<string, OpenAIEffortControlState<TEffort>>,
|
|
56
|
+
key: string,
|
|
57
|
+
): OpenAIEffortControlState<TEffort> {
|
|
58
|
+
const existing = states.get(key);
|
|
59
|
+
if (existing) {
|
|
60
|
+
states.delete(key);
|
|
61
|
+
states.set(key, existing);
|
|
62
|
+
return existing;
|
|
63
|
+
}
|
|
64
|
+
const created = createOpenAIEffortControlState<TEffort>();
|
|
65
|
+
states.set(key, created);
|
|
66
|
+
if (states.size > MAX_EFFORT_CONTROL_STATES) {
|
|
67
|
+
const oldest = states.keys().next().value;
|
|
68
|
+
if (oldest !== undefined) states.delete(oldest);
|
|
69
|
+
}
|
|
70
|
+
return created;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
interface AnchorableItem {
|
|
74
|
+
type?: string | null;
|
|
75
|
+
role?: string;
|
|
76
|
+
id?: string | null;
|
|
77
|
+
status?: string | null;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Fingerprint of the item a transition sits after. Output-only lifecycle
|
|
82
|
+
* fields are excluded: a live response item carries `id`/`status` that the
|
|
83
|
+
* sanitized replay of the same item drops.
|
|
84
|
+
*/
|
|
85
|
+
function effortControlAnchor(input: readonly AnchorableItem[], index: number): string {
|
|
86
|
+
if (index === 0) return "";
|
|
87
|
+
const item = input[index - 1];
|
|
88
|
+
if (!item) return "";
|
|
89
|
+
const { id: _id, status: _status, ...stable } = item;
|
|
90
|
+
return String(Bun.hash(JSON.stringify(stable)));
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
function resetOpenAIEffortControlState(state: OpenAIEffortControlState<string>): void {
|
|
94
|
+
state.baseEffort = undefined;
|
|
95
|
+
state.currentEffort = undefined;
|
|
96
|
+
state.transitions = [];
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Discard the baseline when the request no longer continues the conversation
|
|
101
|
+
* it was captured for: a wire history that shrank or was rewritten under a
|
|
102
|
+
* recorded transition (compaction, branch switch, `/clear`). The next request
|
|
103
|
+
* re-baselines from its own effort, which is what the API asks for after
|
|
104
|
+
* compaction anyway.
|
|
105
|
+
*/
|
|
106
|
+
function syncOpenAIEffortControlState(state: OpenAIEffortControlState<string>, input: readonly AnchorableItem[]): void {
|
|
107
|
+
for (const transition of state.transitions) {
|
|
108
|
+
if (transition.index > input.length || transition.anchor !== effortControlAnchor(input, transition.index)) {
|
|
109
|
+
resetOpenAIEffortControlState(state);
|
|
110
|
+
return;
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Pin the request-level effort to the session baseline and splice pending
|
|
117
|
+
* `configuration_update` items into `input` (mutated in place).
|
|
118
|
+
*
|
|
119
|
+
* `input` is the freshly built transcript for this request, without any
|
|
120
|
+
* `configuration_update` items. `requested` is the wire effort the caller
|
|
121
|
+
* would otherwise send at the request level. Returns the effort to send at the
|
|
122
|
+
* request level (`requested` on the first request, the baseline afterwards).
|
|
123
|
+
*/
|
|
124
|
+
export function planStableOpenAIEffort<TItem extends AnchorableItem, TEffort extends string>(
|
|
125
|
+
state: OpenAIEffortControlState<TEffort>,
|
|
126
|
+
input: Array<TItem | ConfigurationUpdateItem>,
|
|
127
|
+
requested: TEffort,
|
|
128
|
+
): TEffort {
|
|
129
|
+
syncOpenAIEffortControlState(state, input);
|
|
130
|
+
if (state.baseEffort === undefined) {
|
|
131
|
+
state.baseEffort = requested;
|
|
132
|
+
state.currentEffort = requested;
|
|
133
|
+
return requested;
|
|
134
|
+
}
|
|
135
|
+
if (state.currentEffort !== requested) {
|
|
136
|
+
const last = input[input.length - 1];
|
|
137
|
+
const index = last && "role" in last && last.role === "user" ? input.length - 1 : input.length;
|
|
138
|
+
const existing = state.transitions.find(transition => transition.index === index);
|
|
139
|
+
if (existing) {
|
|
140
|
+
existing.effort = requested;
|
|
141
|
+
} else {
|
|
142
|
+
state.transitions.push({ index, anchor: effortControlAnchor(input, index), effort: requested });
|
|
143
|
+
}
|
|
144
|
+
// A change back to the effort already in force at that position is a
|
|
145
|
+
// no-op on the wire; drop it rather than send a redundant item.
|
|
146
|
+
let preceding = state.baseEffort;
|
|
147
|
+
let precedingIndex = -1;
|
|
148
|
+
for (const transition of state.transitions) {
|
|
149
|
+
if (transition.index < index && transition.index > precedingIndex) {
|
|
150
|
+
preceding = transition.effort;
|
|
151
|
+
precedingIndex = transition.index;
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
if (requested === preceding) {
|
|
155
|
+
state.transitions = state.transitions.filter(transition => transition.index !== index);
|
|
156
|
+
}
|
|
157
|
+
state.currentEffort = requested;
|
|
158
|
+
}
|
|
159
|
+
// Splice in ascending order so each insertion offsets only the ones after it.
|
|
160
|
+
state.transitions.sort((a, b) => a.index - b.index);
|
|
161
|
+
let offset = 0;
|
|
162
|
+
for (const transition of state.transitions) {
|
|
163
|
+
input.splice(transition.index + offset, 0, {
|
|
164
|
+
type: "configuration_update",
|
|
165
|
+
reasoning: { effort: transition.effort },
|
|
166
|
+
});
|
|
167
|
+
offset++;
|
|
168
|
+
}
|
|
169
|
+
return state.baseEffort;
|
|
170
|
+
}
|
|
@@ -3013,6 +3013,7 @@ export type ResponseInputItem =
|
|
|
3013
3013
|
| ResponseCustomToolCallOutput
|
|
3014
3014
|
| ResponseCustomToolCall
|
|
3015
3015
|
| ResponseInputItem.CompactionTrigger
|
|
3016
|
+
| ResponseInputItem.ConfigurationUpdate
|
|
3016
3017
|
| ResponseInputItem.ItemReference;
|
|
3017
3018
|
export declare namespace ResponseInputItem {
|
|
3018
3019
|
/**
|
|
@@ -3599,6 +3600,20 @@ export declare namespace ResponseInputItem {
|
|
|
3599
3600
|
*/
|
|
3600
3601
|
type: "compaction_trigger";
|
|
3601
3602
|
}
|
|
3603
|
+
/**
|
|
3604
|
+
* Changes reasoning effort for subsequent responses without touching the
|
|
3605
|
+
* request-level `reasoning.effort` (GPT-6 Astra). Must not be adjacent to
|
|
3606
|
+
* another `configuration_update`.
|
|
3607
|
+
*/
|
|
3608
|
+
interface ConfigurationUpdate {
|
|
3609
|
+
/**
|
|
3610
|
+
* The type of the item. Always `configuration_update`.
|
|
3611
|
+
*/
|
|
3612
|
+
type: "configuration_update";
|
|
3613
|
+
reasoning: {
|
|
3614
|
+
effort: string;
|
|
3615
|
+
};
|
|
3616
|
+
}
|
|
3602
3617
|
/**
|
|
3603
3618
|
* An internal identifier for an item to reference.
|
|
3604
3619
|
*/
|
|
@@ -48,6 +48,11 @@ import {
|
|
|
48
48
|
type OpenAIResponsesToolChoice,
|
|
49
49
|
} from "../utils/tool-choice";
|
|
50
50
|
import { compactGrammarDefinition } from "./grammar";
|
|
51
|
+
import {
|
|
52
|
+
getOpenAIEffortControlState,
|
|
53
|
+
type OpenAIEffortControlState,
|
|
54
|
+
planStableOpenAIEffort,
|
|
55
|
+
} from "./openai-configuration-update";
|
|
51
56
|
import {
|
|
52
57
|
applyOpenAIReasoningEffortFallback,
|
|
53
58
|
clearOpenAIReasoningEffortFallbackState,
|
|
@@ -61,6 +66,7 @@ import {
|
|
|
61
66
|
} from "./openai-reasoning-fallback";
|
|
62
67
|
import type {
|
|
63
68
|
Tool as OpenAITool,
|
|
69
|
+
ReasoningEffort,
|
|
64
70
|
ResponseCreateParamsStreaming,
|
|
65
71
|
ResponseInput,
|
|
66
72
|
ResponseInputContent,
|
|
@@ -197,8 +203,13 @@ interface OpenAIResponsesProviderSessionState
|
|
|
197
203
|
nativeHistoryReplayWarmed: boolean;
|
|
198
204
|
/** Stateful `previous_response_id` chain baselines, keyed by baseUrl/model/session. */
|
|
199
205
|
chains: Map<string, OpenAIResponsesChainState>;
|
|
206
|
+
/** `configuration_update` effort baselines, keyed by baseUrl/model/session. */
|
|
207
|
+
effortControls: Map<string, OpenAIEffortControlState<ResponsesStableEffort>>;
|
|
200
208
|
}
|
|
201
209
|
|
|
210
|
+
/** Wire efforts a `configuration_update` can carry: every real tier, never `none`/null. */
|
|
211
|
+
type ResponsesStableEffort = Exclude<ReasoningEffort, "none" | null>;
|
|
212
|
+
|
|
202
213
|
interface OpenAIResponsesChainState {
|
|
203
214
|
/**
|
|
204
215
|
* Wire params of the last successful turn; never carries
|
|
@@ -224,9 +235,11 @@ function createOpenAIResponsesProviderSessionState(): OpenAIResponsesProviderSes
|
|
|
224
235
|
...reasoningEffortFallbackState,
|
|
225
236
|
nativeHistoryReplayWarmed: false,
|
|
226
237
|
chains: new Map(),
|
|
238
|
+
effortControls: new Map(),
|
|
227
239
|
close: () => {
|
|
228
240
|
state.nativeHistoryReplayWarmed = false;
|
|
229
241
|
state.chains.clear();
|
|
242
|
+
state.effortControls.clear();
|
|
230
243
|
clearOpenAIStrictToolsState(state);
|
|
231
244
|
clearOpenAIReasoningEffortFallbackState(state);
|
|
232
245
|
},
|
|
@@ -1275,6 +1288,7 @@ export function buildParams(
|
|
|
1275
1288
|
if (model.reasoningMode && !options?.forceReasoningOff) {
|
|
1276
1289
|
params.reasoning = { ...params.reasoning, mode: model.reasoningMode };
|
|
1277
1290
|
}
|
|
1291
|
+
applyResponsesStableEffort(model, params, messages, options, providerSessionState);
|
|
1278
1292
|
|
|
1279
1293
|
if (model.compat.isVercelGatewayHost) {
|
|
1280
1294
|
applyVercelResponsesCacheControls(params, model.compat, cacheRetention);
|
|
@@ -1299,6 +1313,33 @@ export function buildParams(
|
|
|
1299
1313
|
return { params, trailingScaffoldingItems, strictToolsApplied };
|
|
1300
1314
|
}
|
|
1301
1315
|
|
|
1316
|
+
/**
|
|
1317
|
+
* Keep the request-level effort byte-stable across a conversation and carry
|
|
1318
|
+
* later changes as `configuration_update` items (GPT-6 Astra). Requires a
|
|
1319
|
+
* routing session id and provider session state to remember the baseline;
|
|
1320
|
+
* without them every request stands alone and sends its own effort.
|
|
1321
|
+
*/
|
|
1322
|
+
function applyResponsesStableEffort(
|
|
1323
|
+
model: Model<"openai-responses">,
|
|
1324
|
+
params: OpenAIResponsesSamplingParams,
|
|
1325
|
+
input: ResponseInput,
|
|
1326
|
+
options: OpenAIResponsesOptions | undefined,
|
|
1327
|
+
providerSessionState: OpenAIResponsesProviderSessionState | undefined,
|
|
1328
|
+
): void {
|
|
1329
|
+
if (!model.compat.supportsConfigurationUpdate || !providerSessionState) return;
|
|
1330
|
+
const reasoning = params.reasoning;
|
|
1331
|
+
if (!reasoning || !("effort" in reasoning)) return;
|
|
1332
|
+
const effort = reasoning.effort;
|
|
1333
|
+
if (effort === undefined || effort === null || effort === "none") return;
|
|
1334
|
+
const sessionId = getOpenAIResponsesRoutingSessionId(options);
|
|
1335
|
+
if (!sessionId) return;
|
|
1336
|
+
const state = getOpenAIEffortControlState(
|
|
1337
|
+
providerSessionState.effortControls,
|
|
1338
|
+
`${model.baseUrl ?? ""}\u0000${model.id}\u0000${sessionId}`,
|
|
1339
|
+
);
|
|
1340
|
+
params.reasoning = { ...reasoning, effort: planStableOpenAIEffort(state, input, effort) };
|
|
1341
|
+
}
|
|
1342
|
+
|
|
1302
1343
|
/**
|
|
1303
1344
|
* Whether this model should get the OpenAI custom-tool grammar variant
|
|
1304
1345
|
* for `apply_patch`. The generated model catalog sets
|
|
@@ -2111,6 +2111,19 @@ function parseResponseReasoningReplayItem(signature: string | undefined): Respon
|
|
|
2111
2111
|
}
|
|
2112
2112
|
}
|
|
2113
2113
|
|
|
2114
|
+
/**
|
|
2115
|
+
* Non-empty `reasoning_text` shipped for a synthesized reasoning item when no
|
|
2116
|
+
* thinking text survived history reconstruction. DeepSeek-family Responses
|
|
2117
|
+
* targets (e.g. opencode-go) reject BOTH a missing reasoning item and one whose
|
|
2118
|
+
* `reasoning_text` is empty — "The reasoning_text in the thinking mode must be
|
|
2119
|
+
* passed back to the API" (#8248 covered the missing case, #10690 the empty
|
|
2120
|
+
* one). The item's presence plus a non-empty payload is what satisfies the
|
|
2121
|
+
* contract; the exact text is immaterial once the source turn's reasoning is
|
|
2122
|
+
* gone. Kept out of `reasoning_content="."`-territory since DeepSeek rejects the
|
|
2123
|
+
* bare-dot synthetic placeholder on the chat-completions path.
|
|
2124
|
+
*/
|
|
2125
|
+
export const SYNTHETIC_REASONING_REPLAY_PLACEHOLDER = "reasoning unavailable";
|
|
2126
|
+
|
|
2114
2127
|
export function convertResponsesAssistantMessage<TApi extends Api>(
|
|
2115
2128
|
assistantMsg: AssistantMessage,
|
|
2116
2129
|
model: Model<TApi>,
|
|
@@ -2268,12 +2281,15 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
|
|
|
2268
2281
|
if (requiresReasoningItem && !reasoningItemEmitted && outputItems.length > 0) {
|
|
2269
2282
|
// Replay the demoted reasoning (already present in `content` as visible
|
|
2270
2283
|
// text) as a structured reasoning item so the thinking-mode continuation
|
|
2271
|
-
// carries the `reasoning_text` the provider requires.
|
|
2272
|
-
//
|
|
2273
|
-
//
|
|
2274
|
-
//
|
|
2275
|
-
//
|
|
2276
|
-
|
|
2284
|
+
// carries the `reasoning_text` the provider requires. When no thinking
|
|
2285
|
+
// text survived reconstruction (source turn minted by another model, or
|
|
2286
|
+
// reasoning dropped by compaction/archive budget) the carried text is
|
|
2287
|
+
// empty — and DeepSeek-family targets reject an empty `reasoning_text`
|
|
2288
|
+
// exactly like a missing item (#10690), so substitute a non-empty
|
|
2289
|
+
// placeholder. The `id` still prefers a surviving upstream item id.
|
|
2290
|
+
const carriedReasoningText = carriedReasoningTexts.join("\n");
|
|
2291
|
+
const reasoningText =
|
|
2292
|
+
carriedReasoningText.length > 0 ? carriedReasoningText : SYNTHETIC_REASONING_REPLAY_PLACEHOLDER;
|
|
2277
2293
|
const reasoningId =
|
|
2278
2294
|
synthesizedReasoningItemId ?? `rs_${Bun.hash(`${model.id}:${msgIndex}:${reasoningText}`).toString(36)}`;
|
|
2279
2295
|
const reasoningItem: ResponseReasoningItem = {
|