@oh-my-pi/pi-ai 18.2.0 → 18.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +35 -0
- package/dist/types/auth-broker/remote-store.d.ts +17 -0
- package/dist/types/auth-gateway/index.d.ts +1 -0
- package/dist/types/auth-gateway/session-state.d.ts +65 -0
- package/dist/types/auth-storage.d.ts +16 -0
- package/dist/types/error/body-error.d.ts +15 -0
- package/dist/types/error/flags.d.ts +16 -0
- package/dist/types/error/index.d.ts +1 -0
- package/dist/types/oneshot-retry.d.ts +6 -0
- package/dist/types/providers/openai-codex/request-transformer.d.ts +27 -0
- package/dist/types/providers/openai-shared.d.ts +20 -3
- package/dist/types/registry/oauth/perplexity.d.ts +1 -7
- package/dist/types/registry/oauth/types.d.ts +8 -0
- package/dist/types/stream.d.ts +2 -0
- package/dist/types/types.d.ts +3 -1
- package/dist/types/usage.d.ts +8 -0
- package/dist/types/utils/block-symbols.d.ts +36 -0
- package/dist/types/utils/openai-http.d.ts +2 -0
- package/dist/types/utils/retry-after.d.ts +2 -0
- package/dist/types/utils/schema/wire.d.ts +4 -5
- package/dist/types/utils.d.ts +9 -0
- package/package.json +6 -6
- package/src/auth-broker/remote-store.ts +73 -8
- package/src/auth-broker/wire-schemas.ts +1 -0
- package/src/auth-gateway/index.ts +1 -0
- package/src/auth-gateway/server.ts +48 -11
- package/src/auth-gateway/session-state.ts +114 -0
- package/src/auth-storage.ts +144 -13
- package/src/error/body-error.ts +310 -0
- package/src/error/flags.ts +63 -13
- package/src/error/index.ts +1 -0
- package/src/error/retryable.ts +2 -0
- package/src/oneshot-retry.ts +13 -3
- package/src/providers/anthropic-messages-server.ts +24 -3
- package/src/providers/anthropic.ts +101 -15
- package/src/providers/cursor.ts +7 -1
- package/src/providers/devin.ts +82 -28
- package/src/providers/openai-chat-server.ts +4 -0
- package/src/providers/openai-codex/request-transformer.ts +36 -0
- package/src/providers/openai-codex-responses.ts +35 -12
- package/src/providers/openai-completions.ts +43 -12
- package/src/providers/openai-reasoning-fallback.ts +6 -6
- package/src/providers/openai-responses-server.ts +2 -1
- package/src/providers/openai-responses.ts +25 -4
- package/src/providers/openai-shared.ts +199 -51
- package/src/registry/oauth/perplexity.ts +94 -28
- package/src/registry/oauth/types.ts +9 -0
- package/src/stream.ts +23 -2
- package/src/types.ts +3 -0
- package/src/usage/claude.ts +33 -0
- package/src/usage/google-antigravity.ts +8 -2
- package/src/usage.ts +3 -0
- package/src/utils/block-symbols.ts +57 -0
- package/src/utils/openai-http.ts +39 -3
- package/src/utils/retry-after.ts +12 -0
- package/src/utils/schema/normalize.ts +3 -3
- package/src/utils/schema/stamps.ts +33 -45
- package/src/utils/schema/wire.ts +9 -7
- package/src/utils.ts +67 -22
|
@@ -237,6 +237,40 @@ function toolOutputKind(type: unknown): ToolCallKind | undefined {
|
|
|
237
237
|
* tool-result child is dropped from the reconstructed history) or when a turn
|
|
238
238
|
* is aborted/crashes after the call streamed but before its result persisted.
|
|
239
239
|
*/
|
|
240
|
+
|
|
241
|
+
/**
|
|
242
|
+
* Sanitize an OpenAI Responses/Codex tool call ID to <= 64 characters and valid charset.
|
|
243
|
+
* Composite IDs with '|' or '\n' have their secondary/item part stripped.
|
|
244
|
+
* Hashing is anchored on the canonical base part so assistant and result composites
|
|
245
|
+
* with different item halves stay identical. Short lossy changes include a hash suffix
|
|
246
|
+
* to preserve collision resistance across distinct IDs.
|
|
247
|
+
*/
|
|
248
|
+
export function sanitizeCodexCallId(rawCallId: string): string {
|
|
249
|
+
if (!rawCallId) return `call_${Bun.hash("empty").toString(36)}`;
|
|
250
|
+
const sep = rawCallId.search(/[\n|]/);
|
|
251
|
+
const base = sep > 0 ? rawCallId.slice(0, sep) : sep === 0 ? rawCallId.slice(1) : rawCallId;
|
|
252
|
+
const sanitized = base.replace(/[^a-zA-Z0-9_-]/g, "_").replace(/_+$/, "");
|
|
253
|
+
if (sanitized.length > 0 && sanitized.length <= 64 && sanitized === base) {
|
|
254
|
+
return sanitized;
|
|
255
|
+
}
|
|
256
|
+
const hash = Bun.hash(base || rawCallId).toString(36);
|
|
257
|
+
const effectiveBase = sanitized.length > 0 ? sanitized : "call";
|
|
258
|
+
const prefixLen = Math.max(0, 63 - hash.length);
|
|
259
|
+
return `${effectiveBase.slice(0, prefixLen)}_${hash}`.slice(0, 64);
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/**
|
|
263
|
+
* In-place mutates the `call_id` property on every input item in the array to conform
|
|
264
|
+
* to the OpenAI Responses/Codex 64-character limit and valid charset constraints.
|
|
265
|
+
*/
|
|
266
|
+
export function sanitizeInputCallIds(input: InputItem[]): void {
|
|
267
|
+
for (const item of input) {
|
|
268
|
+
if (typeof item.call_id === "string") {
|
|
269
|
+
item.call_id = sanitizeCodexCallId(item.call_id);
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
|
|
240
274
|
function repairToolCallPairs(input: InputItem[]): InputItem[] {
|
|
241
275
|
const callKinds = new Map<string, ToolCallKind>();
|
|
242
276
|
const outputKinds = new Map<string, ToolCallKind>();
|
|
@@ -332,6 +366,7 @@ export interface CodexLiteShapedBody {
|
|
|
332
366
|
export function applyCodexResponsesLiteShape(body: CodexLiteShapedBody): void {
|
|
333
367
|
const input = Array.isArray(body.input) ? body.input : [];
|
|
334
368
|
stripImageDetails(input);
|
|
369
|
+
sanitizeInputCallIds(input as InputItem[]);
|
|
335
370
|
body.parallel_tool_calls = false;
|
|
336
371
|
const declaredTools = Array.isArray(body.tools) ? body.tools : [];
|
|
337
372
|
let additionalTools = declaredTools;
|
|
@@ -382,6 +417,7 @@ export async function transformRequestBody(
|
|
|
382
417
|
if (body.input && Array.isArray(body.input)) {
|
|
383
418
|
body.input = filterInput(body.input);
|
|
384
419
|
if (body.input) {
|
|
420
|
+
sanitizeInputCallIds(body.input);
|
|
385
421
|
body.input = repairToolCallPairs(body.input);
|
|
386
422
|
}
|
|
387
423
|
}
|
|
@@ -78,6 +78,7 @@ import {
|
|
|
78
78
|
type ReasoningConfig,
|
|
79
79
|
type RequestBody,
|
|
80
80
|
resolveCodexResponsesLite,
|
|
81
|
+
sanitizeCodexCallId,
|
|
81
82
|
transformRequestBody,
|
|
82
83
|
} from "./openai-codex/request-transformer";
|
|
83
84
|
import { CodexApiError } from "./openai-codex/response-handler";
|
|
@@ -811,6 +812,7 @@ interface CodexOpenItem {
|
|
|
811
812
|
contentIndex: number;
|
|
812
813
|
itemId?: string;
|
|
813
814
|
outputIndex?: number;
|
|
815
|
+
nativeOutputItem?: Record<string, unknown>;
|
|
814
816
|
}
|
|
815
817
|
|
|
816
818
|
class CodexStreamRuntime {
|
|
@@ -840,6 +842,7 @@ class CodexStreamRuntime {
|
|
|
840
842
|
currentItem: CodexEventItem | null = null;
|
|
841
843
|
currentBlock: CodexOutputBlock | null = null;
|
|
842
844
|
nativeOutputItems: Array<Record<string, unknown>> = [];
|
|
845
|
+
nativeOutputEntries: CodexOpenItem[] = [];
|
|
843
846
|
/** Sequential-cutoff summary sections/emitted text, global to the response (indices span reasoning items). */
|
|
844
847
|
cutoffSummaries: SequentialCutoffSummaryState = createSequentialCutoffSummaryState();
|
|
845
848
|
/** Summary deltas buffered while waiting to see whether atomic `.done` events arrive. */
|
|
@@ -876,10 +879,23 @@ class CodexStreamRuntime {
|
|
|
876
879
|
this.currentItem = null;
|
|
877
880
|
this.currentBlock = null;
|
|
878
881
|
this.nativeOutputItems.length = 0;
|
|
882
|
+
this.nativeOutputEntries.length = 0;
|
|
879
883
|
this.pendingSummaryDeltas.clear();
|
|
880
884
|
this.cutoffSummaries = createSequentialCutoffSummaryState();
|
|
881
885
|
}
|
|
882
886
|
|
|
887
|
+
finalizeNativeOutputItems(): Array<Record<string, unknown>> {
|
|
888
|
+
if (this.nativeOutputEntries.length === 0) return this.nativeOutputItems;
|
|
889
|
+
const ordered: Array<Record<string, unknown>> = [];
|
|
890
|
+
for (const entry of this.nativeOutputEntries) {
|
|
891
|
+
if (entry.nativeOutputItem) ordered.push(entry.nativeOutputItem);
|
|
892
|
+
}
|
|
893
|
+
ordered.push(...this.nativeOutputItems);
|
|
894
|
+
this.nativeOutputEntries.length = 0;
|
|
895
|
+
this.nativeOutputItems = ordered;
|
|
896
|
+
return ordered;
|
|
897
|
+
}
|
|
898
|
+
|
|
883
899
|
/**
|
|
884
900
|
* Look up the open item a Codex stream event targets. `item_id` wins because it
|
|
885
901
|
* uniquely identifies a response item; `output_index` covers idless function
|
|
@@ -2195,6 +2211,7 @@ class CodexStreamProcessor {
|
|
|
2195
2211
|
? Math.trunc(rawEvent.output_index)
|
|
2196
2212
|
: undefined;
|
|
2197
2213
|
const entry: CodexOpenItem = { item, block: this.runtime.currentBlock, contentIndex, itemId, outputIndex };
|
|
2214
|
+
this.runtime.nativeOutputEntries.push(entry);
|
|
2198
2215
|
this.runtime.currentEntry = entry;
|
|
2199
2216
|
if (itemId) this.runtime.openItems.set(itemId, entry);
|
|
2200
2217
|
if (outputIndex !== undefined) this.runtime.openItemsByOutputIndex.set(outputIndex, entry);
|
|
@@ -2401,7 +2418,6 @@ class CodexStreamProcessor {
|
|
|
2401
2418
|
if (!rawItem || typeof rawItem !== "object") return;
|
|
2402
2419
|
const item = structuredCloneJSON(rawItem) as CodexEventItem;
|
|
2403
2420
|
if (item.type === "image_generation_call" && item.result) item.status = "completed";
|
|
2404
|
-
runtime.nativeOutputItems.push(item as unknown as Record<string, unknown>);
|
|
2405
2421
|
|
|
2406
2422
|
// Match the finalization to the OPEN ITEM that started this block, not the
|
|
2407
2423
|
// singleton current — interleaved items can finish out of order, so the
|
|
@@ -2410,6 +2426,9 @@ class CodexStreamProcessor {
|
|
|
2410
2426
|
// routes `output_item.done` to the block that received `output_item.added`.
|
|
2411
2427
|
const itemId = "id" in item && typeof item.id === "string" ? item.id : "";
|
|
2412
2428
|
const entry = (itemId ? runtime.openItems.get(itemId) : null) ?? runtime.openItemForEvent(rawEvent);
|
|
2429
|
+
const nativeOutputItem = item as unknown as Record<string, unknown>;
|
|
2430
|
+
if (entry) entry.nativeOutputItem = nativeOutputItem;
|
|
2431
|
+
else runtime.nativeOutputItems.push(nativeOutputItem);
|
|
2413
2432
|
const block = entry?.block ?? null;
|
|
2414
2433
|
const contentIndex = entry?.contentIndex ?? output.content.length - 1;
|
|
2415
2434
|
|
|
@@ -2543,16 +2562,19 @@ class CodexStreamProcessor {
|
|
|
2543
2562
|
resetCodexWebSocketAppendState(state);
|
|
2544
2563
|
} else {
|
|
2545
2564
|
state.lastRequest = structuredCloneJSON(runtime.requestBodyForState);
|
|
2565
|
+
const nativeOutputItems = runtime.finalizeNativeOutputItems();
|
|
2546
2566
|
const replayableResponseItems = sanitizeOpenAIResponsesAssistantHistoryItemsForReplay(
|
|
2547
|
-
structuredCloneJSON(
|
|
2567
|
+
structuredCloneJSON(nativeOutputItems),
|
|
2548
2568
|
);
|
|
2549
|
-
if (responseId && replayableResponseItems) {
|
|
2569
|
+
if (responseId && replayableResponseItems && replayableResponseItems.length === nativeOutputItems.length) {
|
|
2550
2570
|
state.lastResponseId = responseId;
|
|
2551
2571
|
state.lastResponseItems = replayableResponseItems;
|
|
2552
2572
|
state.canAppend = rawEvent.type === "response.done" || rawEvent.type === "response.completed";
|
|
2553
2573
|
} else {
|
|
2554
|
-
//
|
|
2555
|
-
|
|
2574
|
+
// No response id, or replay sanitization dropped an item the server
|
|
2575
|
+
// still holds. Sanitization is 1:1-or-fewer, so either case makes the
|
|
2576
|
+
// append baseline untrustworthy; next turn must replay in full.
|
|
2577
|
+
resetCodexWebSocketAppendState(state);
|
|
2556
2578
|
}
|
|
2557
2579
|
}
|
|
2558
2580
|
}
|
|
@@ -2967,7 +2989,10 @@ class CodexStreamProcessor {
|
|
|
2967
2989
|
throw new CodexProviderStreamError("Codex response failed", false);
|
|
2968
2990
|
}
|
|
2969
2991
|
|
|
2970
|
-
output.providerPayload = createOpenAIResponsesHistoryPayload(
|
|
2992
|
+
output.providerPayload = createOpenAIResponsesHistoryPayload(
|
|
2993
|
+
this.model.provider,
|
|
2994
|
+
this.runtime.finalizeNativeOutputItems(),
|
|
2995
|
+
);
|
|
2971
2996
|
output.duration = performance.now() - this.startTime;
|
|
2972
2997
|
if (completion.firstTokenTime) {
|
|
2973
2998
|
output.ttft = completion.firstTokenTime - this.startTime;
|
|
@@ -4528,16 +4553,14 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
|
|
|
4528
4553
|
const messages: ResponseInput = [];
|
|
4529
4554
|
|
|
4530
4555
|
const normalizeToolCallId = (id: string): string => {
|
|
4531
|
-
|
|
4532
|
-
const [callId, itemId] = id.
|
|
4533
|
-
const
|
|
4534
|
-
let sanitizedItemId = itemId.replace(/[^a-zA-Z0-9_-]/g, "_");
|
|
4556
|
+
const sep = id.search(/[\n|]/);
|
|
4557
|
+
const [callId, itemId] = sep > 0 ? [id.slice(0, sep), id.slice(sep + 1)] : [id, undefined];
|
|
4558
|
+
const normalizedCallId = sanitizeCodexCallId(callId);
|
|
4559
|
+
let sanitizedItemId = (itemId ?? Bun.hash(id).toString(36)).replace(/[^a-zA-Z0-9_-]/g, "_");
|
|
4535
4560
|
if (!sanitizedItemId.startsWith("fc")) {
|
|
4536
4561
|
sanitizedItemId = `fc_${sanitizedItemId}`;
|
|
4537
4562
|
}
|
|
4538
|
-
let normalizedCallId = sanitizedCallId.length > 64 ? sanitizedCallId.slice(0, 64) : sanitizedCallId;
|
|
4539
4563
|
let normalizedItemId = sanitizedItemId.length > 64 ? sanitizedItemId.slice(0, 64) : sanitizedItemId;
|
|
4540
|
-
normalizedCallId = normalizedCallId.replace(/_+$/, "");
|
|
4541
4564
|
normalizedItemId = normalizedItemId.replace(/_+$/, "");
|
|
4542
4565
|
return `${normalizedCallId}|${normalizedItemId}`;
|
|
4543
4566
|
};
|
|
@@ -677,7 +677,7 @@ const streamOpenAICompletionsOnce = (
|
|
|
677
677
|
(async () => {
|
|
678
678
|
const startTime = performance.now();
|
|
679
679
|
let firstTokenTime: number | undefined;
|
|
680
|
-
const policy = resolveOpenAICompatForRequest(model, options);
|
|
680
|
+
const policy = resolveOpenAICompatForRequest(model, options, Boolean(context.tools?.length));
|
|
681
681
|
|
|
682
682
|
const output: AssistantMessage = createInitialResponsesAssistantMessage(model.api, model.provider, model.id);
|
|
683
683
|
let rawRequestDump: RawHttpRequestDump | undefined;
|
|
@@ -765,14 +765,17 @@ const streamOpenAICompletionsOnce = (
|
|
|
765
765
|
const builtParams = buildParams(model, context, options, effectiveToolStrictModeOverride);
|
|
766
766
|
appliedStrictTools = builtParams.strictToolsApplied;
|
|
767
767
|
let params = builtParams.params;
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
params.model
|
|
772
|
-
|
|
773
|
-
const requestReasoningEffortFallback =
|
|
774
|
-
|
|
775
|
-
|
|
768
|
+
// Tool-triggered suppression is a hard wire constraint; cached
|
|
769
|
+
// enabled-effort negotiation must not overwrite its `none`.
|
|
770
|
+
const reasoningEffortFallbackKey = builtParams.reasoningEffortFallbackAllowed
|
|
771
|
+
? createOpenAIReasoningEffortFallbackKey("chat-completions", trimmedBaseUrl, params.model)
|
|
772
|
+
: undefined;
|
|
773
|
+
const requestReasoningEffortFallback =
|
|
774
|
+
reasoningEffortFallbackKey === undefined
|
|
775
|
+
? undefined
|
|
776
|
+
: requestReasoningEffortFallbacks.has(reasoningEffortFallbackKey)
|
|
777
|
+
? requestReasoningEffortFallbacks.get(reasoningEffortFallbackKey)
|
|
778
|
+
: getOpenAIReasoningEffortFallback(providerSessionState, reasoningEffortFallbackKey);
|
|
776
779
|
if (requestReasoningEffortFallback !== undefined) {
|
|
777
780
|
applyOpenAIReasoningEffortFallback(params, requestReasoningEffortFallback);
|
|
778
781
|
}
|
|
@@ -1166,6 +1169,21 @@ const streamOpenAICompletionsOnce = (
|
|
|
1166
1169
|
});
|
|
1167
1170
|
for await (const chunk of terminalAwareStream) {
|
|
1168
1171
|
if (!chunk || typeof chunk !== "object") continue;
|
|
1172
|
+
// Rate-limit/overload bodies sent inside an HTTP 200 stream (Azure,
|
|
1173
|
+
// LiteLLM-style aggregators, some gates) arrive as an `error` member or
|
|
1174
|
+
// a bare `{ code, status }` chunk. This probe runs first: the legacy
|
|
1175
|
+
// stream-error guard below turns *any* object `error` member into a
|
|
1176
|
+
// statusless `ProviderResponseError`, so if it went first no throttle
|
|
1177
|
+
// envelope would ever reach the in-band classifier.
|
|
1178
|
+
//
|
|
1179
|
+
// Invariants (body-error.ts): the status is read only from error
|
|
1180
|
+
// `status`/`code` fields and restricted to 429/5xx — never derived from
|
|
1181
|
+
// prose, so a body mentioning 401/403 stays out of the auth lane — and a
|
|
1182
|
+
// synthesized message is never opaque, so an unreadable body cannot burn
|
|
1183
|
+
// a credential. Envelopes that are not a recognised throttle return
|
|
1184
|
+
// `undefined` and keep their pre-existing handling.
|
|
1185
|
+
const inBand = AIError.createInBandProviderError(chunk);
|
|
1186
|
+
if (inBand) throw inBand;
|
|
1169
1187
|
const streamError = createOpenAICompletionsStreamError(chunk, model.provider);
|
|
1170
1188
|
if (streamError) throw streamError;
|
|
1171
1189
|
|
|
@@ -1570,12 +1588,14 @@ function createRequestSetup(
|
|
|
1570
1588
|
function resolveOpenAICompatForRequest(
|
|
1571
1589
|
model: Model<"openai-completions">,
|
|
1572
1590
|
options: OpenAICompletionsOptions | undefined,
|
|
1591
|
+
hasTools: boolean,
|
|
1573
1592
|
): OpenAICompatPolicy {
|
|
1574
1593
|
return resolveOpenAICompatPolicy(model, {
|
|
1575
1594
|
endpoint: "chat-completions",
|
|
1576
1595
|
reasoning: options?.reasoning,
|
|
1577
1596
|
disableReasoning: options?.disableReasoning,
|
|
1578
1597
|
toolChoice: mapToOpenAICompletionsToolChoice(options?.toolChoice),
|
|
1598
|
+
hasTools,
|
|
1579
1599
|
});
|
|
1580
1600
|
}
|
|
1581
1601
|
|
|
@@ -1678,8 +1698,9 @@ function buildParams(
|
|
|
1678
1698
|
params: OpenAICompletionsParams;
|
|
1679
1699
|
toolStrictMode: AppliedToolStrictMode;
|
|
1680
1700
|
strictToolsApplied: boolean;
|
|
1701
|
+
reasoningEffortFallbackAllowed: boolean;
|
|
1681
1702
|
} {
|
|
1682
|
-
const initialPolicy = resolveOpenAICompatForRequest(model, options);
|
|
1703
|
+
const initialPolicy = resolveOpenAICompatForRequest(model, options, Boolean(context.tools?.length));
|
|
1683
1704
|
const initialCompat = initialPolicy.compat as ResolvedOpenAICompat;
|
|
1684
1705
|
const cacheRetention = resolveCacheRetention(options?.cacheRetention);
|
|
1685
1706
|
|
|
@@ -1833,6 +1854,7 @@ function buildParams(
|
|
|
1833
1854
|
reasoning: options?.reasoning,
|
|
1834
1855
|
disableReasoning: options?.disableReasoning,
|
|
1835
1856
|
toolChoice: params.tool_choice,
|
|
1857
|
+
hasTools: Array.isArray(params.tools) && params.tools.length > 0,
|
|
1836
1858
|
});
|
|
1837
1859
|
const compat = finalPolicy.compat as ResolvedOpenAICompat;
|
|
1838
1860
|
const messages = convertMessages(model, context, compat);
|
|
@@ -1857,7 +1879,11 @@ function buildParams(
|
|
|
1857
1879
|
}
|
|
1858
1880
|
applyChatCompletionsToolStream(params, model, compat);
|
|
1859
1881
|
|
|
1860
|
-
applyChatCompletionsReasoningParams(params, model, compat, {
|
|
1882
|
+
applyChatCompletionsReasoningParams(params, model, compat, {
|
|
1883
|
+
...options,
|
|
1884
|
+
toolChoice: params.tool_choice,
|
|
1885
|
+
hasTools: Array.isArray(params.tools) && params.tools.length > 0,
|
|
1886
|
+
});
|
|
1861
1887
|
dropOpenRouterKimiForcedToolReasoning(params, model, finalPolicy);
|
|
1862
1888
|
|
|
1863
1889
|
applyOpenAIGatewayRouting(params, compat, cacheRetention !== "none");
|
|
@@ -1867,7 +1893,12 @@ function buildParams(
|
|
|
1867
1893
|
});
|
|
1868
1894
|
applyOpenAIChatCompletionsPromptCachePolicy(params, model, options);
|
|
1869
1895
|
|
|
1870
|
-
return {
|
|
1896
|
+
return {
|
|
1897
|
+
params,
|
|
1898
|
+
toolStrictMode,
|
|
1899
|
+
strictToolsApplied,
|
|
1900
|
+
reasoningEffortFallbackAllowed: finalPolicy.reasoning.disableReason !== "tools",
|
|
1901
|
+
};
|
|
1871
1902
|
}
|
|
1872
1903
|
|
|
1873
1904
|
export function parseChunkUsage(
|
|
@@ -187,7 +187,7 @@ function collectMessageParts(error: unknown, captured: CapturedHttpErrorResponse
|
|
|
187
187
|
* the same rejection with `Supported values are: …`, so value lists count too.
|
|
188
188
|
*/
|
|
189
189
|
const REASONING_EFFORT_FIELD_PATTERN =
|
|
190
|
-
/reasoning[_. ]effort|reasoning value|(?:valid|supported|allowed) (?:levels?|values?)/i;
|
|
190
|
+
/reasoning[_. ]?effort|reasoning value|(?:valid|supported|allowed) (?:levels?|values?)/i;
|
|
191
191
|
|
|
192
192
|
function mentionsReasoningEffort(error: unknown, captured: CapturedHttpErrorResponse | undefined): boolean {
|
|
193
193
|
const param = capturedStringField(captured, "param");
|
|
@@ -224,17 +224,17 @@ interface EffortRejectionSignal {
|
|
|
224
224
|
rejectedMatches: boolean;
|
|
225
225
|
}
|
|
226
226
|
|
|
227
|
-
const EFFORT_FIELD_PATTERN = /reasoning[_. ]effort|reasoning value/i;
|
|
227
|
+
const EFFORT_FIELD_PATTERN = /reasoning[_. ]?effort|reasoning value/i;
|
|
228
228
|
const ALLOWED_LEVELS_PATTERN = /(?:valid|supported|allowed) levels?/i;
|
|
229
229
|
|
|
230
230
|
/** Fielded rejection verdicts in any word order: verdict-first, field-first, or bare mention plus verdict. */
|
|
231
231
|
function messageCarriesEffortVerdict(message: string): boolean {
|
|
232
232
|
return (
|
|
233
|
-
/invalid[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(message) ||
|
|
234
|
-
/(?:reasoning[_. ]effort|reasoning value)[^\n]*(?:invalid|unsupported|not supported|not permitted|must be|expected|unknown|unexpected|unrecognized)/i.test(
|
|
233
|
+
/invalid[^\n]*(?:reasoning[_. ]?effort|reasoning value)/i.test(message) ||
|
|
234
|
+
/(?:reasoning[_. ]?effort|reasoning value)[^\n]*(?:invalid|unsupported|not supported|not permitted|must be|expected|unknown|unexpected|unrecognized)/i.test(
|
|
235
235
|
message,
|
|
236
236
|
) ||
|
|
237
|
-
/(?:unsupported|not supported|not permitted|unknown|unexpected|unrecognized|extra)[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(
|
|
237
|
+
/(?:unsupported|not supported|not permitted|unknown|unexpected|unrecognized|extra)[^\n]*(?:reasoning[_. ]?effort|reasoning value)/i.test(
|
|
238
238
|
message,
|
|
239
239
|
)
|
|
240
240
|
);
|
|
@@ -367,7 +367,7 @@ function nearestEnabledReasoningFallback(currentEffort: string, allowed: Set<str
|
|
|
367
367
|
* supported`).
|
|
368
368
|
*/
|
|
369
369
|
const TEMPLATE_KWARG_EFFORT_PATTERN =
|
|
370
|
-
/chat_template_kwargs[^\n]{0,120}reasoning[_. ]effort|reasoning[_. ]effort[^\n]{0,120}chat_template_kwargs/i;
|
|
370
|
+
/chat_template_kwargs[^\n]{0,120}reasoning[_. ]?effort|reasoning[_. ]?effort[^\n]{0,120}chat_template_kwargs/i;
|
|
371
371
|
const FIELD_REJECTION_PATTERN =
|
|
372
372
|
/invalid|unsupported|not supported|not permitted|unknown|unexpected|unrecognized|rejected|extra input/i;
|
|
373
373
|
|
|
@@ -43,7 +43,7 @@ import {
|
|
|
43
43
|
type OpenAIResponsesTool,
|
|
44
44
|
openaiResponsesRequestSchema,
|
|
45
45
|
} from "./openai-responses-server-schema";
|
|
46
|
-
import { encodeTextSignatureV1, parseTextSignature } from "./openai-shared";
|
|
46
|
+
import { coerceNullMessageContentInPlace, encodeTextSignatureV1, parseTextSignature } from "./openai-shared";
|
|
47
47
|
|
|
48
48
|
export type { ParsedRequest };
|
|
49
49
|
|
|
@@ -365,6 +365,7 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
|
|
|
365
365
|
// `resolvePromptCacheKey` call further down.
|
|
366
366
|
|
|
367
367
|
rejectUnsupportedExplicitPromptCacheFields(body);
|
|
368
|
+
coerceNullMessageContentInPlace(isObj(body) ? body.input : undefined);
|
|
368
369
|
const data = openaiResponsesRequestSchema(body);
|
|
369
370
|
if (data instanceof type.errors) {
|
|
370
371
|
throw new AIError.ValidationError(`openai-responses: ${data.summary}`);
|
|
@@ -422,6 +422,7 @@ const streamOpenAIResponsesOnce = (
|
|
|
422
422
|
let rawRequestDump: RawHttpRequestDump | undefined;
|
|
423
423
|
let chainState: OpenAIResponsesChainState | undefined;
|
|
424
424
|
let sentPreviousResponseId: string | undefined;
|
|
425
|
+
let lastSubmittedRequestWasFullReplay: boolean | undefined;
|
|
425
426
|
const abortTracker = createAbortSourceTracker(options?.signal);
|
|
426
427
|
const firstEventTimeoutAbortError = new AIError.StreamTimeoutError(OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE);
|
|
427
428
|
const { requestAbortController, requestSignal } = abortTracker;
|
|
@@ -552,6 +553,7 @@ const streamOpenAIResponsesOnce = (
|
|
|
552
553
|
typeof requestParams.model === "string" ? requestParams.model : model.id,
|
|
553
554
|
);
|
|
554
555
|
activeRequestParams = requestParams;
|
|
556
|
+
lastSubmittedRequestWasFullReplay = requestParams.previous_response_id === undefined;
|
|
555
557
|
let requestTimeout: NodeJS.Timeout | undefined;
|
|
556
558
|
if (requestTimeoutMs !== undefined) {
|
|
557
559
|
requestTimeout = setTimeout(
|
|
@@ -576,6 +578,9 @@ const streamOpenAIResponsesOnce = (
|
|
|
576
578
|
copilotCacheKey,
|
|
577
579
|
copilotCacheSnapshot,
|
|
578
580
|
),
|
|
581
|
+
shouldRetryResponse: (response, bodyText) =>
|
|
582
|
+
!AIError.isRequestBodyReadTimeout(response.status, bodyText) ||
|
|
583
|
+
lastSubmittedRequestWasFullReplay !== true,
|
|
579
584
|
// Transient 408/429/5xx get Retry-After-aware transport
|
|
580
585
|
// retries; the first-event watchdog aborts `requestSignal`,
|
|
581
586
|
// so retries cannot extend the caller's deadline.
|
|
@@ -878,7 +883,7 @@ const streamOpenAIResponsesOnce = (
|
|
|
878
883
|
: activeParams,
|
|
879
884
|
);
|
|
880
885
|
chainState.lastPromptCacheBreakpointPolicy = promptCacheBreakpointPolicy;
|
|
881
|
-
if (output.responseId) {
|
|
886
|
+
if (output.responseId && replayableResponseItems.length === nativeOutputItems.length) {
|
|
882
887
|
chainState.lastResponseId = output.responseId;
|
|
883
888
|
chainState.lastResponseItems = replayableResponseItems;
|
|
884
889
|
chainState.canAppend = true;
|
|
@@ -886,8 +891,12 @@ const streamOpenAIResponsesOnce = (
|
|
|
886
891
|
// full-context success must not mask categorical rejection.
|
|
887
892
|
if (sentPreviousResponseId) chainState.staleFailures = 0;
|
|
888
893
|
} else {
|
|
889
|
-
//
|
|
894
|
+
// No response id, or replay sanitization dropped an item the server
|
|
895
|
+
// still holds. Sanitization is 1:1-or-fewer, so either case makes the
|
|
896
|
+
// append baseline untrustworthy; next turn must replay in full.
|
|
890
897
|
chainState.canAppend = false;
|
|
898
|
+
chainState.lastResponseId = undefined;
|
|
899
|
+
chainState.lastResponseItems = undefined;
|
|
891
900
|
}
|
|
892
901
|
}
|
|
893
902
|
} else if (chainState) {
|
|
@@ -926,6 +935,9 @@ const streamOpenAIResponsesOnce = (
|
|
|
926
935
|
output.errorStatus = result.status;
|
|
927
936
|
output.errorId = result.id;
|
|
928
937
|
output.errorMessage = result.message;
|
|
938
|
+
if (AIError.isRequestBodyReadTimeout(result.status, result.message) && lastSubmittedRequestWasFullReplay) {
|
|
939
|
+
output.requestBodyReadTimeoutFullReplay = true;
|
|
940
|
+
}
|
|
929
941
|
// Some providers via OpenRouter include extra details here.
|
|
930
942
|
const rawMetadata = (error as { error?: { metadata?: { raw?: string } } })?.error?.metadata?.raw;
|
|
931
943
|
if (rawMetadata) output.errorMessage += `\n${rawMetadata}`;
|
|
@@ -1153,6 +1165,11 @@ export function buildParams(
|
|
|
1153
1165
|
});
|
|
1154
1166
|
const strictResponsesPairing = policy.tools.strictResponsesPairing;
|
|
1155
1167
|
const shouldReplayNativeHistory = providerSessionState?.nativeHistoryReplayWarmed ?? true;
|
|
1168
|
+
// Filtering native reasoning must not be undone by reconstruction when the
|
|
1169
|
+
// target also rejects synthetic items (Muse on OpenRouter). Unfiltered targets
|
|
1170
|
+
// retain required text/placeholder replay, including DeepSeek's #10690 fallback.
|
|
1171
|
+
const canReconstructReasoningReplay =
|
|
1172
|
+
!policy.reasoning.filterReasoningHistory || policy.reasoning.allowsSyntheticReasoningContentForToolCalls;
|
|
1156
1173
|
const messages = buildResponsesInput({
|
|
1157
1174
|
model,
|
|
1158
1175
|
context,
|
|
@@ -1164,9 +1181,13 @@ export function buildParams(
|
|
|
1164
1181
|
},
|
|
1165
1182
|
includeThinkingSignatures: shouldReplayNativeHistory && !policy.reasoning.filterReasoningHistory,
|
|
1166
1183
|
requiresReasoningReplayForAllTurns:
|
|
1167
|
-
policy.reasoning.enabled &&
|
|
1184
|
+
policy.reasoning.enabled &&
|
|
1185
|
+
policy.reasoning.requiresReasoningContentForAllAssistantTurns &&
|
|
1186
|
+
canReconstructReasoningReplay,
|
|
1168
1187
|
requiresReasoningReplayForToolCalls:
|
|
1169
|
-
policy.reasoning.enabled &&
|
|
1188
|
+
policy.reasoning.enabled &&
|
|
1189
|
+
policy.reasoning.requiresReasoningContentForToolCalls &&
|
|
1190
|
+
canReconstructReasoningReplay,
|
|
1170
1191
|
repairOrphanOutputs: true,
|
|
1171
1192
|
});
|
|
1172
1193
|
|