@oh-my-pi/pi-ai 18.0.4 → 18.0.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +4 -1
- package/dist/types/providers/anthropic.d.ts +2 -5
- package/dist/types/providers/azure-openai-responses.d.ts +4 -1
- package/dist/types/providers/openai-completions.d.ts +2 -4
- package/dist/types/providers/openai-responses.d.ts +3 -4
- package/dist/types/providers/openai-shared.d.ts +2 -1
- package/dist/types/registry/deepinfra.d.ts +6 -0
- package/dist/types/registry/oauth/openrouter.d.ts +36 -0
- package/dist/types/registry/openrouter.d.ts +7 -6
- package/dist/types/registry/registry.d.ts +10 -0
- package/dist/types/registry/yolo-auto.d.ts +15 -0
- package/dist/types/utils/empty-completion-retry.d.ts +14 -5
- package/package.json +5 -5
- package/src/error/flags.ts +1 -0
- package/src/providers/anthropic.ts +6 -7
- package/src/providers/azure-openai-responses.ts +21 -9
- package/src/providers/ollama.ts +4 -2
- package/src/providers/openai-codex-responses.ts +101 -28
- package/src/providers/openai-completions.ts +26 -8
- package/src/providers/openai-responses.ts +9 -7
- package/src/providers/openai-shared.ts +28 -6
- package/src/providers/vision-guard.ts +9 -1
- package/src/registry/deepinfra.ts +24 -0
- package/src/registry/oauth/openrouter.ts +120 -0
- package/src/registry/openrouter.ts +16 -20
- package/src/registry/registry.ts +4 -0
- package/src/registry/yolo-auto.ts +30 -0
- package/src/utils/empty-completion-retry.ts +75 -41
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,27 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.0.5] - 2026-08-25
|
|
6
|
+
|
|
7
|
+
### Breaking Changes
|
|
8
|
+
|
|
9
|
+
- Renamed the exported stream-retry helper from `withEmptyCompletionRetry` to `withReplaySafeStreamRetry` and added retry policy options for empty completions and provider errors. Consumers using the old helper must migrate.
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- Added browser-based Sign in with OpenRouter using OAuth PKCE, while retaining support for pasted OpenRouter API keys and redirect URLs for remote sessions.
|
|
14
|
+
- Added `/login` API-key authentication for DeepInfra and Yolo-Auto, including validation against each provider before the credentials are accepted.
|
|
15
|
+
|
|
16
|
+
### Fixed
|
|
17
|
+
|
|
18
|
+
- Fixed DeepSeek vision models from losing image input while keeping image parts stripped for text-only DeepSeek endpoints.
|
|
19
|
+
- Fixed OpenAI-compatible gateways that report uppercase completion reasons such as `STOP` or `MAX_TOKENS`; these are now classified correctly, including mapping `MAX_TOKENS` to a length limit.
|
|
20
|
+
- Fixed provider message-count limit errors being treated as unrecoverable payload errors instead of recoverable context overflows.
|
|
21
|
+
- Improved Codex WebSocket continuations so rate limits, throttling, and compatible mode changes preserve valid response continuations instead of unnecessarily replaying the full context.
|
|
22
|
+
- Fixed Codex WebSocket cleanup failures caused by already-closed sockets.
|
|
23
|
+
- Added safe retries for transient mid-stream socket closures across OpenAI Responses, Chat Completions, Azure OpenAI Responses, and Codex SSE when no replay-unsafe output has been emitted.
|
|
24
|
+
- Fixed usage and cost reporting for OpenAI-compatible gateways backed by Vertex AI or Gemini by recognizing cached prompt tokens reported through `cachedContentTokenCount`.
|
|
25
|
+
|
|
5
26
|
## [18.0.4] - 2026-08-24
|
|
6
27
|
|
|
7
28
|
### Fixed
|
package/README.md
CHANGED
|
@@ -60,6 +60,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
|
|
|
60
60
|
- **NVIDIA** (requires `NVIDIA_API_KEY`)
|
|
61
61
|
- **NanoGPT** (requires `NANO_GPT_API_KEY`)
|
|
62
62
|
- **Novita** (requires `NOVITA_API_KEY`)
|
|
63
|
+
- **DeepInfra** (requires `DEEPINFRA_API_KEY`)
|
|
63
64
|
- **Hugging Face Inference**
|
|
64
65
|
- **xAI**
|
|
65
66
|
- **Venice** (requires `VENICE_API_KEY`)
|
|
@@ -943,6 +944,7 @@ In Node.js environments, you can set environment variables to avoid passing API
|
|
|
943
944
|
| NVIDIA | `NVIDIA_API_KEY` |
|
|
944
945
|
| NanoGPT | `NANO_GPT_API_KEY` |
|
|
945
946
|
| Novita | `NOVITA_API_KEY` |
|
|
947
|
+
| DeepInfra | `DEEPINFRA_API_KEY` |
|
|
946
948
|
| Venice | `VENICE_API_KEY` |
|
|
947
949
|
| Moonshot | `MOONSHOT_API_KEY` |
|
|
948
950
|
| xAI | `XAI_API_KEY` |
|
|
@@ -983,6 +985,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
|
|
|
983
985
|
- NVIDIA: `https://integrate.api.nvidia.com/v1`
|
|
984
986
|
- NanoGPT: `https://nano-gpt.com/api/v1`
|
|
985
987
|
- Novita: `https://api.novita.ai/openai/v1`
|
|
988
|
+
- DeepInfra: `https://api.deepinfra.com/v1/openai`
|
|
986
989
|
- Hugging Face Inference: `https://router.huggingface.co/v1`
|
|
987
990
|
- Venice: `https://api.venice.ai/api/v1`
|
|
988
991
|
- Xiaomi MiMo: `https://api.xiaomimimo.com/anthropic`
|
|
@@ -1084,7 +1087,7 @@ Credentials are saved to `agent.db` in the agent directory. `/login qianfan` ope
|
|
|
1084
1087
|
|
|
1085
1088
|
`login` supports OAuth providers (Anthropic, OpenAI Codex, GitHub Copilot, Gemini CLI, Antigravity) and API-key onboarding flows.
|
|
1086
1089
|
|
|
1087
|
-
For the current API-key onboarding flows, the library covers Together, Moonshot, Qianfan, NVIDIA, NanoGPT, Novita, Hugging Face, Venice, Xiaomi, vLLM, LiteLLM, Cloudflare AI Gateway, Qwen Portal, and Ollama Cloud. Ollama remains the local runtime integration; set `OLLAMA_API_KEY` only when your local or self-hosted deployment enforces bearer auth.
|
|
1090
|
+
For the current API-key onboarding flows, the library covers Together, Moonshot, Qianfan, NVIDIA, NanoGPT, Novita, DeepInfra, Hugging Face, Venice, Xiaomi, vLLM, LiteLLM, Cloudflare AI Gateway, Qwen Portal, and Ollama Cloud. Ollama remains the local runtime integration; set `OLLAMA_API_KEY` only when your local or self-hosted deployment enforces bearer auth.
|
|
1088
1091
|
|
|
1089
1092
|
### Programmatic OAuth
|
|
1090
1093
|
|
|
@@ -218,11 +218,8 @@ export declare function isInvalidThinkingSignatureError(message: string): boolea
|
|
|
218
218
|
*/
|
|
219
219
|
export declare function maybeAddReplayUnsignedThinkingHint(model: Model<"anthropic-messages">, message: string): string;
|
|
220
220
|
/**
|
|
221
|
-
* Public entry:
|
|
222
|
-
*
|
|
223
|
-
* stall the agent loop). The inner attempt keeps its own provider-failure retry
|
|
224
|
-
* loop; this layer only re-issues a fresh request on an empty success. Shared
|
|
225
|
-
* with the OpenAI-completions provider via `withEmptyCompletionRetry`.
|
|
221
|
+
* Public entry: retry benign empty completions before they reach the agent
|
|
222
|
+
* loop. The inner attempt owns Anthropic provider-failure retries.
|
|
226
223
|
*/
|
|
227
224
|
export declare const streamAnthropic: StreamFunction<"anthropic-messages">;
|
|
228
225
|
export type AnthropicSystemBlock = {
|
|
@@ -12,6 +12,9 @@ export interface AzureOpenAIResponsesOptions extends StreamOptions {
|
|
|
12
12
|
disableReasoning?: boolean;
|
|
13
13
|
}
|
|
14
14
|
/**
|
|
15
|
-
*
|
|
15
|
+
* Retries transient Azure stream failures only before assistant output commits
|
|
16
|
+
* the attempt. The unsupported explicit prompt-cache config is rejected
|
|
17
|
+
* synchronously here — callers of the direct entrypoint get the immediate
|
|
18
|
+
* `ConfigurationError` rather than a stream whose `.result()` rejects later.
|
|
16
19
|
*/
|
|
17
20
|
export declare const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses">;
|
|
@@ -38,10 +38,8 @@ export interface OpenAICompletionsOptions extends StreamOptions {
|
|
|
38
38
|
promptCache?: OpenAIPromptCacheOptions;
|
|
39
39
|
}
|
|
40
40
|
/**
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
* "stop"` and no usage, which would otherwise stall the agent loop. Shared with
|
|
44
|
-
* the Anthropic provider via `withEmptyCompletionRetry`.
|
|
41
|
+
* Retries benign empty completions and transient provider failures only before
|
|
42
|
+
* assistant output commits the attempt.
|
|
45
43
|
*/
|
|
46
44
|
export declare const streamOpenAICompletions: StreamFunction<"openai-completions">;
|
|
47
45
|
export declare function parseChunkUsage(rawUsage: object, model: Model<"openai-completions">, premiumRequests: number | undefined): AssistantMessage["usage"];
|
|
@@ -106,10 +106,9 @@ type OpenAIResponsesSamplingParams = ResponseCreateParamsStreaming & {
|
|
|
106
106
|
cache_ttl?: "5m" | "1h";
|
|
107
107
|
};
|
|
108
108
|
/**
|
|
109
|
-
* Public entry:
|
|
110
|
-
*
|
|
111
|
-
*
|
|
112
|
-
* Anthropic providers via `withEmptyCompletionRetry`.
|
|
109
|
+
* Public entry: retry benign empty completions before they reach the agent
|
|
110
|
+
* loop. Transient stream failures are retried inside the attempt so stateful
|
|
111
|
+
* Responses request metadata remains stable.
|
|
113
112
|
*/
|
|
114
113
|
export declare const streamOpenAIResponses: StreamFunction<"openai-responses">;
|
|
115
114
|
export declare function buildParams(model: Model<"openai-responses">, context: Context, options: OpenAIResponsesOptions | undefined, providerSessionState: OpenAIResponsesProviderSessionState | undefined, strictToolsScope?: OpenAIStrictToolsScope, disableStrictToolsOverride?: boolean, statefulCacheBaseline?: ResponseInput): {
|
|
@@ -234,6 +234,7 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
|
|
|
234
234
|
preserve_thinking?: boolean;
|
|
235
235
|
chat_template_kwargs?: {
|
|
236
236
|
enable_thinking?: boolean;
|
|
237
|
+
thinking?: boolean;
|
|
237
238
|
preserve_thinking?: boolean;
|
|
238
239
|
reasoning_effort?: string;
|
|
239
240
|
};
|
|
@@ -684,4 +685,4 @@ export declare function buildResponsesDeltaInput<TItem extends ResponseInputItem
|
|
|
684
685
|
input?: TItem[];
|
|
685
686
|
} | undefined, previousResponseItems: readonly TItem[] | undefined, current: {
|
|
686
687
|
input?: TItem[];
|
|
687
|
-
}): TItem[] | null;
|
|
688
|
+
}, additionalTopLevelExcludeMap?: Readonly<Record<string, boolean>>): TItem[] | null;
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
export declare const loginDeepinfra: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
2
|
+
export declare const deepinfraProvider: {
|
|
3
|
+
id: "deepinfra";
|
|
4
|
+
name: string;
|
|
5
|
+
login: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
6
|
+
};
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Sign in with OpenRouter (OAuth PKCE).
|
|
3
|
+
*
|
|
4
|
+
* No client registration: the S256 PKCE challenge is the only proof of
|
|
5
|
+
* identity. The browser authorizes at `https://openrouter.ai/auth`, redirects
|
|
6
|
+
* back to the loopback callback with `?code=`, and the code + verifier
|
|
7
|
+
* exchange at `/api/v1/auth/keys` mints a durable `sk-or-…` API key.
|
|
8
|
+
*
|
|
9
|
+
* OpenRouter never echoes a `state` parameter (the redirect only appends
|
|
10
|
+
* `?code=` to the callback URL), so the flow runs with an empty expected
|
|
11
|
+
* state — callback-state validation is disabled and the PKCE verifier binds
|
|
12
|
+
* the exchange instead.
|
|
13
|
+
*
|
|
14
|
+
* The manual-input race doubles as the API-key paste path: pasted `sk-or-…`
|
|
15
|
+
* keys skip the PKCE exchange and are validated against `/api/v1/auth/key`
|
|
16
|
+
* (the canonical "who am I" endpoint — `/api/v1/models` returns 200 for any
|
|
17
|
+
* bearer), so one login entry covers both browser sign-in and key paste.
|
|
18
|
+
*/
|
|
19
|
+
import type { FetchImpl } from "../../types.js";
|
|
20
|
+
import { OAuthCallbackFlow } from "./callback-server.js";
|
|
21
|
+
import type { OAuthController, OAuthCredentials } from "./types.js";
|
|
22
|
+
/** Exchange an authorization code + PKCE verifier for a durable OpenRouter API key. */
|
|
23
|
+
export declare function exchangeOpenRouterCode(code: string, codeVerifier: string, fetchImpl?: FetchImpl): Promise<string>;
|
|
24
|
+
export declare class OpenRouterOAuthFlow extends OAuthCallbackFlow {
|
|
25
|
+
#private;
|
|
26
|
+
constructor(ctrl: OAuthController);
|
|
27
|
+
/** OpenRouter never echoes `state`; empty state disables callback-state validation. */
|
|
28
|
+
generateState(): string;
|
|
29
|
+
generateAuthUrl(_state: string, redirectUri: string): Promise<{
|
|
30
|
+
url: string;
|
|
31
|
+
instructions?: string;
|
|
32
|
+
}>;
|
|
33
|
+
exchangeToken(code: string): Promise<OAuthCredentials>;
|
|
34
|
+
}
|
|
35
|
+
/** Log in with Sign in with OpenRouter (PKCE); mints a durable API key. */
|
|
36
|
+
export declare function loginOpenRouterOAuth(ctrl: OAuthController): Promise<OAuthCredentials>;
|
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
import type { OAuthLoginCallbacks } from "./oauth/types.js";
|
|
2
|
-
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
2
|
+
/**
|
|
3
|
+
* OpenRouter login: Sign in with OpenRouter (OAuth PKCE) that mints a durable
|
|
4
|
+
* `sk-or-…` API key, with the manual-input race accepting a pasted existing
|
|
5
|
+
* key (validated via `/api/v1/auth/key`). Either path resolves to a plain API
|
|
6
|
+
* key string, stored as an `api_key` credential.
|
|
7
7
|
*/
|
|
8
|
-
export declare const loginOpenRouter: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
9
8
|
export declare const openrouterProvider: {
|
|
10
9
|
readonly id: "openrouter";
|
|
11
10
|
readonly name: "OpenRouter";
|
|
12
11
|
readonly login: (cb: OAuthLoginCallbacks) => Promise<string>;
|
|
12
|
+
readonly callbackPort: 54549;
|
|
13
|
+
readonly pasteCodeFlow: true;
|
|
13
14
|
};
|
|
@@ -82,6 +82,10 @@ declare const ALL: ({
|
|
|
82
82
|
readonly name: "Cursor (Claude, GPT, etc.)";
|
|
83
83
|
readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<import("./oauth/index.js").OAuthCredentials>;
|
|
84
84
|
readonly refreshToken: (credentials: import("./oauth/index.js").OAuthCredentials) => Promise<import("./oauth/index.js").OAuthCredentials>;
|
|
85
|
+
} | {
|
|
86
|
+
id: "deepinfra";
|
|
87
|
+
name: string;
|
|
88
|
+
login: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
85
89
|
} | {
|
|
86
90
|
readonly id: "deepseek";
|
|
87
91
|
readonly name: "DeepSeek";
|
|
@@ -254,6 +258,8 @@ declare const ALL: ({
|
|
|
254
258
|
readonly id: "openrouter";
|
|
255
259
|
readonly name: "OpenRouter";
|
|
256
260
|
readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string>;
|
|
261
|
+
readonly callbackPort: 54549;
|
|
262
|
+
readonly pasteCodeFlow: true;
|
|
257
263
|
} | {
|
|
258
264
|
readonly id: "parallel";
|
|
259
265
|
readonly name: "Parallel";
|
|
@@ -342,6 +348,10 @@ declare const ALL: ({
|
|
|
342
348
|
readonly id: "xiaomi-token-plan-sgp";
|
|
343
349
|
readonly name: "Xiaomi Token Plan (Singapore)";
|
|
344
350
|
readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string>;
|
|
351
|
+
} | {
|
|
352
|
+
readonly id: "yolo-auto";
|
|
353
|
+
readonly name: "Yolo-Auto";
|
|
354
|
+
readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string>;
|
|
345
355
|
} | {
|
|
346
356
|
readonly id: "zai";
|
|
347
357
|
readonly name: "Z.AI (GLM Coding Plan)";
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import type { OAuthLoginCallbacks } from "./oauth/types.js";
|
|
2
|
+
/**
|
|
3
|
+
* Yolo-Auto login flow (API key paste, validated via `/v1/models`).
|
|
4
|
+
*
|
|
5
|
+
* Yolo-Auto is a flat-rate OpenAI-compatible API. `GET /v1/models` 401s any
|
|
6
|
+
* missing or invalid bearer (verified against the live endpoint), so it doubles
|
|
7
|
+
* as the canonical "who am I" check — the same role OpenRouter's `/auth/key`
|
|
8
|
+
* plays there.
|
|
9
|
+
*/
|
|
10
|
+
export declare const loginYoloAuto: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
11
|
+
export declare const yoloAutoProvider: {
|
|
12
|
+
readonly id: "yolo-auto";
|
|
13
|
+
readonly name: "Yolo-Auto";
|
|
14
|
+
readonly login: (cb: OAuthLoginCallbacks) => Promise<string>;
|
|
15
|
+
};
|
|
@@ -8,15 +8,24 @@ export declare const EMPTY_COMPLETION_BASE_DELAY_MS = 500;
|
|
|
8
8
|
* — or one that only ever produced thinking — is the "empty response" failure.
|
|
9
9
|
*/
|
|
10
10
|
export declare function hasVisibleAssistantContent(message: AssistantMessage): boolean;
|
|
11
|
-
interface
|
|
11
|
+
interface StreamRetryOptions {
|
|
12
12
|
signal?: AbortSignal;
|
|
13
13
|
providerRetryWait?: (delayMs: number, signal?: AbortSignal) => Promise<void>;
|
|
14
14
|
acceptEmptyResponse?: boolean;
|
|
15
15
|
}
|
|
16
|
+
/** Controls which replay-safe provider results may issue a fresh request. */
|
|
17
|
+
export interface ReplaySafeStreamRetryPolicy {
|
|
18
|
+
/** Retry benign terminal stops that contain no visible output. */
|
|
19
|
+
retryEmptyCompletion?: boolean;
|
|
20
|
+
/** Retry transient provider errors before output is committed. */
|
|
21
|
+
retryProviderErrors?: boolean;
|
|
22
|
+
/** Maximum transient provider-error retries; empty completions keep their shared fixed budget. */
|
|
23
|
+
maxProviderErrorRetries?: number;
|
|
24
|
+
}
|
|
16
25
|
/**
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
26
|
+
* Re-issues a fresh provider request only while the current attempt remains
|
|
27
|
+
* replay-safe. Buffered pre-output events from discarded attempts never reach
|
|
28
|
+
* consumers.
|
|
20
29
|
*/
|
|
21
|
-
export declare function
|
|
30
|
+
export declare function withReplaySafeStreamRetry<M, O extends StreamRetryOptions>(model: M, context: Context, options: O | undefined, attempt: (model: M, context: Context, options?: O) => AssistantMessageEventStream, policy: ReplaySafeStreamRetryPolicy): AssistantMessageEventStream;
|
|
22
31
|
export {};
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-ai",
|
|
4
|
-
"version": "18.0.
|
|
4
|
+
"version": "18.0.5",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": "Stencil Labs, Inc.",
|
|
@@ -37,10 +37,10 @@
|
|
|
37
37
|
"fmt": "biome format --write ."
|
|
38
38
|
},
|
|
39
39
|
"dependencies": {
|
|
40
|
-
"@oh-my-pi/omptype": "18.0.
|
|
41
|
-
"@oh-my-pi/pi-catalog": "18.0.
|
|
42
|
-
"@oh-my-pi/pi-utils": "18.0.
|
|
43
|
-
"@oh-my-pi/pi-wire": "18.0.
|
|
40
|
+
"@oh-my-pi/omptype": "18.0.5",
|
|
41
|
+
"@oh-my-pi/pi-catalog": "18.0.5",
|
|
42
|
+
"@oh-my-pi/pi-utils": "18.0.5",
|
|
43
|
+
"@oh-my-pi/pi-wire": "18.0.5"
|
|
44
44
|
},
|
|
45
45
|
"devDependencies": {
|
|
46
46
|
"@types/bun": "^1.3.14"
|
package/src/error/flags.ts
CHANGED
|
@@ -99,6 +99,7 @@ const CONTEXT_OVERFLOW_EVIDENCE_PATTERNS = [
|
|
|
99
99
|
/model_context_window_exceeded/i, // z.ai non-standard finish_reason surfaced as error text
|
|
100
100
|
/prompt filled the context window/i, // Ollama OpenAI-compatible empty length completion
|
|
101
101
|
/exceeds the limit of \d+ tokens?\b/i,
|
|
102
|
+
/chat history exceeds the \d+-message limit/i, // Provider message-count cap
|
|
102
103
|
] as const;
|
|
103
104
|
// Numeric limit pattern — also matches media budgets, so must never veto a payload flag (#9235).
|
|
104
105
|
const GENERIC_LIMIT_OVERFLOW_PATTERN = /exceeds the limit of \d+/i;
|
|
@@ -54,7 +54,7 @@ import {
|
|
|
54
54
|
kStreamingLastParseLen,
|
|
55
55
|
kStreamingPartialJson,
|
|
56
56
|
} from "../utils/block-symbols";
|
|
57
|
-
import {
|
|
57
|
+
import { withReplaySafeStreamRetry } from "../utils/empty-completion-retry";
|
|
58
58
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
59
59
|
import { isFoundryEnabled } from "../utils/foundry";
|
|
60
60
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
@@ -2873,14 +2873,13 @@ const streamAnthropicOnce = (
|
|
|
2873
2873
|
};
|
|
2874
2874
|
|
|
2875
2875
|
/**
|
|
2876
|
-
* Public entry:
|
|
2877
|
-
*
|
|
2878
|
-
* stall the agent loop). The inner attempt keeps its own provider-failure retry
|
|
2879
|
-
* loop; this layer only re-issues a fresh request on an empty success. Shared
|
|
2880
|
-
* with the OpenAI-completions provider via `withEmptyCompletionRetry`.
|
|
2876
|
+
* Public entry: retry benign empty completions before they reach the agent
|
|
2877
|
+
* loop. The inner attempt owns Anthropic provider-failure retries.
|
|
2881
2878
|
*/
|
|
2882
2879
|
export const streamAnthropic: StreamFunction<"anthropic-messages"> = (model, context, options) =>
|
|
2883
|
-
|
|
2880
|
+
withReplaySafeStreamRetry(model, context, options, streamAnthropicOnce, {
|
|
2881
|
+
retryEmptyCompletion: true,
|
|
2882
|
+
});
|
|
2884
2883
|
|
|
2885
2884
|
export type AnthropicSystemBlock = {
|
|
2886
2885
|
type: "text";
|
|
@@ -13,6 +13,7 @@ import type {
|
|
|
13
13
|
} from "../types";
|
|
14
14
|
import { resolveCacheRetention } from "../utils";
|
|
15
15
|
import { createAbortSourceTracker } from "../utils/abort";
|
|
16
|
+
import { withReplaySafeStreamRetry } from "../utils/empty-completion-retry";
|
|
16
17
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
17
18
|
import type { RawHttpRequestDump } from "../utils/http-inspector";
|
|
18
19
|
import {
|
|
@@ -76,19 +77,12 @@ type AzureOpenAIResponsesSamplingParams = ResponseCreateParamsStreaming & {
|
|
|
76
77
|
repetition_penalty?: number;
|
|
77
78
|
};
|
|
78
79
|
|
|
79
|
-
/**
|
|
80
|
-
|
|
81
|
-
*/
|
|
82
|
-
export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"> = (
|
|
80
|
+
/** Runs one Azure OpenAI Responses request and decodes its event stream. */
|
|
81
|
+
const streamAzureOpenAIResponsesOnce = (
|
|
83
82
|
model: Model<"azure-openai-responses">,
|
|
84
83
|
context: Context,
|
|
85
84
|
options?: AzureOpenAIResponsesOptions,
|
|
86
85
|
): AssistantMessageEventStream => {
|
|
87
|
-
if (options?.promptCache?.mode === "explicit" && resolveCacheRetention(options.cacheRetention) !== "none") {
|
|
88
|
-
throw new AIError.ConfigurationError(
|
|
89
|
-
`OpenAI explicit prompt caching is unsupported for ${model.provider}/${model.id}; Azure Responses does not emit explicit cache controls.`,
|
|
90
|
-
);
|
|
91
|
-
}
|
|
92
86
|
const stream = new AssistantMessageEventStream();
|
|
93
87
|
|
|
94
88
|
// Start async processing
|
|
@@ -265,6 +259,24 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
265
259
|
return stream;
|
|
266
260
|
};
|
|
267
261
|
|
|
262
|
+
/**
|
|
263
|
+
* Retries transient Azure stream failures only before assistant output commits
|
|
264
|
+
* the attempt. The unsupported explicit prompt-cache config is rejected
|
|
265
|
+
* synchronously here — callers of the direct entrypoint get the immediate
|
|
266
|
+
* `ConfigurationError` rather than a stream whose `.result()` rejects later.
|
|
267
|
+
*/
|
|
268
|
+
export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"> = (model, context, options) => {
|
|
269
|
+
if (options?.promptCache?.mode === "explicit" && resolveCacheRetention(options.cacheRetention) !== "none") {
|
|
270
|
+
throw new AIError.ConfigurationError(
|
|
271
|
+
`OpenAI explicit prompt caching is unsupported for ${model.provider}/${model.id}; Azure Responses does not emit explicit cache controls.`,
|
|
272
|
+
);
|
|
273
|
+
}
|
|
274
|
+
return withReplaySafeStreamRetry(model, context, options, streamAzureOpenAIResponsesOnce, {
|
|
275
|
+
retryProviderErrors: true,
|
|
276
|
+
maxProviderErrorRetries: 1,
|
|
277
|
+
});
|
|
278
|
+
};
|
|
279
|
+
|
|
268
280
|
function resolveAzureConfig(
|
|
269
281
|
model: Model<"azure-openai-responses">,
|
|
270
282
|
options?: AzureOpenAIResponsesOptions,
|
package/src/providers/ollama.ts
CHANGED
|
@@ -16,7 +16,7 @@ import type {
|
|
|
16
16
|
} from "../types";
|
|
17
17
|
import { normalizeSystemPrompts } from "../utils";
|
|
18
18
|
import { clearStreamingPartialJson, kStreamingPartialJson } from "../utils/block-symbols";
|
|
19
|
-
import {
|
|
19
|
+
import { withReplaySafeStreamRetry } from "../utils/empty-completion-retry";
|
|
20
20
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
21
21
|
import type { CapturedHttpErrorResponse, RawHttpRequestDump } from "../utils/http-inspector";
|
|
22
22
|
import {
|
|
@@ -785,4 +785,6 @@ const streamOllamaOnce = (
|
|
|
785
785
|
|
|
786
786
|
/** Retry EOS-only Ollama completions before the agent loop sees an empty stop. */
|
|
787
787
|
export const streamOllama: StreamFunction<"ollama-chat"> = (model, context, options) =>
|
|
788
|
-
|
|
788
|
+
withReplaySafeStreamRetry(model, context, options, streamOllamaOnce, {
|
|
789
|
+
retryEmptyCompletion: true,
|
|
790
|
+
});
|
|
@@ -54,6 +54,7 @@ import {
|
|
|
54
54
|
stripOpenAIResponsesComputerLinkedReasoningIdsForReplay,
|
|
55
55
|
} from "../utils";
|
|
56
56
|
import { clearStreamingPartialJson, kStreamingLastParseLen, kStreamingPartialJson } from "../utils/block-symbols";
|
|
57
|
+
import { hasVisibleAssistantContent } from "../utils/empty-completion-retry";
|
|
57
58
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
58
59
|
import { escapeHarmonyControlTokens, isHarmonyDialectModel } from "../utils/harmony-leak";
|
|
59
60
|
import type { RawHttpRequestDump } from "../utils/http-inspector";
|
|
@@ -458,6 +459,7 @@ export interface OpenAICodexWebSocketDebugStats {
|
|
|
458
459
|
type CodexWebSocketSessionState = {
|
|
459
460
|
disableWebsocket: boolean;
|
|
460
461
|
lastRequest?: RequestBody;
|
|
462
|
+
/** Last completed response; an in-progress response cannot replace the retry baseline. */
|
|
461
463
|
lastResponseId?: string;
|
|
462
464
|
lastResponseItems?: InputItem[];
|
|
463
465
|
canAppend: boolean;
|
|
@@ -1019,14 +1021,6 @@ class CodexStreamRuntime {
|
|
|
1019
1021
|
const input = (rawEvent as { input?: string }).input;
|
|
1020
1022
|
if (typeof input === "string") finalizeCustomToolCallInputDone(entry.block, input);
|
|
1021
1023
|
}
|
|
1022
|
-
|
|
1023
|
-
handleResponseCreated(rawEvent: Record<string, unknown>): void {
|
|
1024
|
-
const response = (rawEvent as { response?: { id?: string } }).response;
|
|
1025
|
-
const state = this.websocketState;
|
|
1026
|
-
if (state && this.transport === "websocket" && typeof response?.id === "string" && response.id.length > 0) {
|
|
1027
|
-
state.lastResponseId = response.id;
|
|
1028
|
-
}
|
|
1029
|
-
}
|
|
1030
1024
|
}
|
|
1031
1025
|
|
|
1032
1026
|
interface CodexWhitespaceToolCallArgumentsDeltaState {
|
|
@@ -1047,6 +1041,7 @@ interface CodexStreamFailureContext {
|
|
|
1047
1041
|
output: AssistantMessage;
|
|
1048
1042
|
options: OpenAICodexResponsesOptions | undefined;
|
|
1049
1043
|
requestContext: CodexRequestContext;
|
|
1044
|
+
runtime?: CodexStreamRuntime;
|
|
1050
1045
|
startTime: number;
|
|
1051
1046
|
firstTokenTime?: number;
|
|
1052
1047
|
}
|
|
@@ -2000,6 +1995,11 @@ const CODEX_STALE_PREVIOUS_RESPONSE_CODES: Record<string, true> = {
|
|
|
2000
1995
|
codex_previous_response_stale: true,
|
|
2001
1996
|
};
|
|
2002
1997
|
|
|
1998
|
+
const CODEX_APPEND_PRESERVING_REJECTION_CODES: Record<string, true> = {
|
|
1999
|
+
rate_limit_exceeded: true,
|
|
2000
|
+
slow_down: true,
|
|
2001
|
+
};
|
|
2002
|
+
|
|
2003
2003
|
function isCodexStalePreviousResponseError(error: unknown): boolean {
|
|
2004
2004
|
if (!(error instanceof Error)) return false;
|
|
2005
2005
|
if (
|
|
@@ -2021,9 +2021,22 @@ function isCodexStalePreviousResponseError(error: unknown): boolean {
|
|
|
2021
2021
|
);
|
|
2022
2022
|
}
|
|
2023
2023
|
|
|
2024
|
+
function shouldPreserveCodexWebSocketAppendState(context: CodexStreamFailureContext, error: unknown): boolean {
|
|
2025
|
+
if (!(error instanceof CodexProviderStreamError) || !error.code) return false;
|
|
2026
|
+
const state = context.requestContext.websocketState;
|
|
2027
|
+
return (
|
|
2028
|
+
Object.hasOwn(CODEX_APPEND_PRESERVING_REJECTION_CODES, error.code.toLowerCase()) &&
|
|
2029
|
+
context.runtime?.transport === "websocket" &&
|
|
2030
|
+
state?.canAppend === true &&
|
|
2031
|
+
state.lastRequest !== undefined &&
|
|
2032
|
+
state.lastResponseId !== undefined &&
|
|
2033
|
+
state.lastResponseItems !== undefined
|
|
2034
|
+
);
|
|
2035
|
+
}
|
|
2036
|
+
|
|
2024
2037
|
async function handleCodexStreamFailure(context: CodexStreamFailureContext, error: unknown): Promise<AssistantMessage> {
|
|
2025
2038
|
const { output } = context;
|
|
2026
|
-
if (context.requestContext.websocketState) {
|
|
2039
|
+
if (context.requestContext.websocketState && !shouldPreserveCodexWebSocketAppendState(context, error)) {
|
|
2027
2040
|
resetCodexWebSocketAppendState(context.requestContext.websocketState);
|
|
2028
2041
|
context.requestContext.websocketState.modelsEtag = undefined;
|
|
2029
2042
|
}
|
|
@@ -2296,11 +2309,6 @@ class CodexStreamProcessor {
|
|
|
2296
2309
|
return firstTokenTime;
|
|
2297
2310
|
}
|
|
2298
2311
|
|
|
2299
|
-
if (eventType === "response.created") {
|
|
2300
|
-
this.runtime.handleResponseCreated(rawEvent);
|
|
2301
|
-
return firstTokenTime;
|
|
2302
|
-
}
|
|
2303
|
-
|
|
2304
2312
|
if (eventType === "response.completed" || eventType === "response.done" || eventType === "response.incomplete") {
|
|
2305
2313
|
this.#handleResponseCompleted(rawEvent);
|
|
2306
2314
|
return firstTokenTime;
|
|
@@ -2779,16 +2787,67 @@ class CodexStreamProcessor {
|
|
|
2779
2787
|
return true;
|
|
2780
2788
|
}
|
|
2781
2789
|
|
|
2790
|
+
/**
|
|
2791
|
+
* Emit balancing `*_end` events for every block that opened (pushed a
|
|
2792
|
+
* `*_start`) but never committed visible content, so a retry that resets
|
|
2793
|
+
* `output` and replays cannot leave the consumer with an orphaned
|
|
2794
|
+
* `text_start`/`thinking_start` from the abandoned attempt. Only reachable
|
|
2795
|
+
* blocks are empty text/reasoning ones — a committed tool/text block blocks
|
|
2796
|
+
* the retry upstream.
|
|
2797
|
+
*/
|
|
2798
|
+
#closeOpenBlocksForReplay(): void {
|
|
2799
|
+
const { runtime, output, stream } = this;
|
|
2800
|
+
const open = new Set<CodexOpenItem>(runtime.openItems.values());
|
|
2801
|
+
for (const entry of runtime.openItemsByOutputIndex.values()) open.add(entry);
|
|
2802
|
+
if (runtime.currentEntry) open.add(runtime.currentEntry);
|
|
2803
|
+
for (const entry of open) {
|
|
2804
|
+
const block = entry.block;
|
|
2805
|
+
if (block?.type === "thinking") {
|
|
2806
|
+
stream.push({
|
|
2807
|
+
type: "thinking_end",
|
|
2808
|
+
contentIndex: entry.contentIndex,
|
|
2809
|
+
content: block.thinking,
|
|
2810
|
+
partial: output,
|
|
2811
|
+
});
|
|
2812
|
+
} else if (block?.type === "text") {
|
|
2813
|
+
stream.push({ type: "text_end", contentIndex: entry.contentIndex, content: block.text, partial: output });
|
|
2814
|
+
}
|
|
2815
|
+
}
|
|
2816
|
+
}
|
|
2817
|
+
|
|
2782
2818
|
async #tryRetryProviderError(error: unknown): Promise<boolean> {
|
|
2819
|
+
const retryable =
|
|
2820
|
+
error instanceof CodexProviderStreamError
|
|
2821
|
+
? error.retryable
|
|
2822
|
+
: AIError.isProviderRetryableError(error, { provider: this.model.provider });
|
|
2823
|
+
// A leading `response.output_item.added` opens an empty block and emits only
|
|
2824
|
+
// a `*_start` before any delta; that is replay-safe. But once any text or
|
|
2825
|
+
// thinking delta has streamed — including a whitespace-only
|
|
2826
|
+
// `output_text.delta`, which still reaches consumers as `text_delta` — or a
|
|
2827
|
+
// visible tool/image block exists, replaying would duplicate/reorder those
|
|
2828
|
+
// already-delivered events, so treat the attempt as committed. Emptiness is
|
|
2829
|
+
// measured by the streamed block length (whitespace counts), not by visible
|
|
2830
|
+
// final content.
|
|
2831
|
+
const streamedContent = this.output.content.some(
|
|
2832
|
+
block =>
|
|
2833
|
+
(block.type === "text" && block.text.length > 0) ||
|
|
2834
|
+
(block.type === "thinking" && block.thinking.length > 0),
|
|
2835
|
+
);
|
|
2783
2836
|
if (
|
|
2784
|
-
!
|
|
2785
|
-
this.output
|
|
2837
|
+
!retryable ||
|
|
2838
|
+
hasVisibleAssistantContent(this.output) ||
|
|
2839
|
+
streamedContent ||
|
|
2840
|
+
!this.runtime.canSafelyReplayWebsocketOverSse ||
|
|
2786
2841
|
this.runtime.providerRetryAttempt >= CODEX_MAX_RETRIES ||
|
|
2787
2842
|
this.options?.signal?.aborted
|
|
2788
2843
|
) {
|
|
2789
2844
|
return false;
|
|
2790
2845
|
}
|
|
2791
2846
|
|
|
2847
|
+
// A leading `output_item.added` already pushed a `*_start` for the (empty)
|
|
2848
|
+
// open block; balance it with the matching end before the reset+replay so
|
|
2849
|
+
// consumers never see an orphaned start from the abandoned attempt.
|
|
2850
|
+
this.#closeOpenBlocksForReplay();
|
|
2792
2851
|
this.runtime.providerRetryAttempt += 1;
|
|
2793
2852
|
const websocketState = this.requestContext.websocketState;
|
|
2794
2853
|
if (websocketState) {
|
|
@@ -3463,10 +3522,14 @@ function recordCodexTurnUsageDiagnostics(
|
|
|
3463
3522
|
CODEX_DEBUG && logger.debug("[codex] codex turn diagnostics", { diagnostics: state.stats.lastTurn });
|
|
3464
3523
|
}
|
|
3465
3524
|
|
|
3525
|
+
const CODEX_CHAIN_TOP_LEVEL_EXCLUDE_MAP = {
|
|
3526
|
+
service_tier: true,
|
|
3527
|
+
};
|
|
3528
|
+
|
|
3466
3529
|
/**
|
|
3467
|
-
* Shape the next websocket turn's request body: when the session's
|
|
3468
|
-
*
|
|
3469
|
-
*
|
|
3530
|
+
* Shape the next websocket turn's request body: when the session's strict
|
|
3531
|
+
* history prefix is intact and request options other than the per-turn
|
|
3532
|
+
* service_tier match, chain via previous_response_id + delta-only input;
|
|
3470
3533
|
* replay the full transcript. SSE requests never chain — the HTTP endpoint's
|
|
3471
3534
|
* request schema has no `previous_response_id` (codex-rs carries it only on
|
|
3472
3535
|
* websocket `response.create` frames) and strict gateway validators 400 it
|
|
@@ -3478,7 +3541,12 @@ function buildCodexChainedRequestBody(
|
|
|
3478
3541
|
): RequestBody {
|
|
3479
3542
|
const chainable = state?.canAppend === true;
|
|
3480
3543
|
const appendInput = chainable
|
|
3481
|
-
? buildResponsesDeltaInput(
|
|
3544
|
+
? buildResponsesDeltaInput(
|
|
3545
|
+
state.lastRequest,
|
|
3546
|
+
state.lastResponseItems,
|
|
3547
|
+
requestBody,
|
|
3548
|
+
CODEX_CHAIN_TOP_LEVEL_EXCLUDE_MAP,
|
|
3549
|
+
)
|
|
3482
3550
|
: null;
|
|
3483
3551
|
if (appendInput && appendInput.length > 0 && state?.lastResponseId) {
|
|
3484
3552
|
return { ...requestBody, previous_response_id: state.lastResponseId, input: appendInput };
|
|
@@ -3599,14 +3667,19 @@ class CodexWebSocketConnection {
|
|
|
3599
3667
|
}
|
|
3600
3668
|
|
|
3601
3669
|
close(reason = "done"): void {
|
|
3602
|
-
|
|
3603
|
-
this.#socket &&
|
|
3604
|
-
(this.#socket.readyState === WebSocket.OPEN || this.#socket.readyState === WebSocket.CONNECTING)
|
|
3605
|
-
) {
|
|
3606
|
-
this.#socket.close(1000, reason);
|
|
3607
|
-
}
|
|
3670
|
+
const socket = this.#socket;
|
|
3608
3671
|
this.#socket = null;
|
|
3609
3672
|
this.#stopHeartbeat();
|
|
3673
|
+
if (!socket || (socket.readyState !== WebSocket.OPEN && socket.readyState !== WebSocket.CONNECTING)) return;
|
|
3674
|
+
try {
|
|
3675
|
+
socket.close(1000, reason);
|
|
3676
|
+
} catch (error) {
|
|
3677
|
+
CODEX_DEBUG &&
|
|
3678
|
+
logger.debug("[codex] codex websocket close failed", {
|
|
3679
|
+
error: error instanceof Error ? error.message : String(error),
|
|
3680
|
+
reason,
|
|
3681
|
+
});
|
|
3682
|
+
}
|
|
3610
3683
|
}
|
|
3611
3684
|
|
|
3612
3685
|
async connect(signal?: AbortSignal): Promise<void> {
|
|
@@ -3634,7 +3707,7 @@ class CodexWebSocketConnection {
|
|
|
3634
3707
|
if (signal) signal.removeEventListener("abort", onAbort);
|
|
3635
3708
|
};
|
|
3636
3709
|
const onAbort = () => {
|
|
3637
|
-
|
|
3710
|
+
this.close("aborted");
|
|
3638
3711
|
if (!settled) {
|
|
3639
3712
|
settled = true;
|
|
3640
3713
|
clearPending();
|
|
@@ -3650,7 +3723,7 @@ class CodexWebSocketConnection {
|
|
|
3650
3723
|
}
|
|
3651
3724
|
if (!settled) {
|
|
3652
3725
|
timeout = setTimeout(() => {
|
|
3653
|
-
|
|
3726
|
+
this.close("connect-timeout");
|
|
3654
3727
|
if (!settled) {
|
|
3655
3728
|
settled = true;
|
|
3656
3729
|
clearPending();
|