@oh-my-pi/pi-ai 18.0.3 → 18.0.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +4 -1
- package/dist/types/auth-storage.d.ts +2 -2
- package/dist/types/error/flags.d.ts +13 -0
- package/dist/types/providers/anthropic.d.ts +2 -5
- package/dist/types/providers/azure-openai-responses.d.ts +4 -1
- package/dist/types/providers/cursor.d.ts +1 -0
- package/dist/types/providers/openai-completions.d.ts +2 -4
- package/dist/types/providers/openai-responses.d.ts +3 -4
- package/dist/types/providers/openai-shared.d.ts +2 -1
- package/dist/types/registry/deepinfra.d.ts +6 -0
- package/dist/types/registry/oauth/openrouter.d.ts +36 -0
- package/dist/types/registry/openrouter.d.ts +7 -6
- package/dist/types/registry/registry.d.ts +10 -0
- package/dist/types/registry/yolo-auto.d.ts +15 -0
- package/dist/types/utils/empty-completion-retry.d.ts +14 -5
- package/package.json +5 -5
- package/src/auth-storage.ts +14 -7
- package/src/error/flags.ts +147 -15
- package/src/oneshot-retry.ts +1 -0
- package/src/providers/anthropic.ts +6 -7
- package/src/providers/azure-openai-responses.ts +21 -9
- package/src/providers/cursor.ts +43 -21
- package/src/providers/ollama.ts +4 -2
- package/src/providers/openai-chat-server.ts +24 -10
- package/src/providers/openai-codex-responses.ts +101 -28
- package/src/providers/openai-completions.ts +26 -8
- package/src/providers/openai-responses.ts +9 -7
- package/src/providers/openai-shared.ts +28 -6
- package/src/providers/vision-guard.ts +9 -1
- package/src/registry/deepinfra.ts +24 -0
- package/src/registry/oauth/openrouter.ts +120 -0
- package/src/registry/openrouter.ts +16 -20
- package/src/registry/registry.ts +4 -0
- package/src/registry/yolo-auto.ts +30 -0
- package/src/stream.ts +2 -2
- package/src/utils/empty-completion-retry.ts +75 -41
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,36 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.0.5] - 2026-08-25
|
|
6
|
+
|
|
7
|
+
### Breaking Changes
|
|
8
|
+
|
|
9
|
+
- Renamed the exported stream-retry helper from `withEmptyCompletionRetry` to `withReplaySafeStreamRetry` and added retry policy options for empty completions and provider errors. Consumers using the old helper must migrate.
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- Added browser-based Sign in with OpenRouter using OAuth PKCE, while retaining support for pasted OpenRouter API keys and redirect URLs for remote sessions.
|
|
14
|
+
- Added `/login` API-key authentication for DeepInfra and Yolo-Auto, including validation against each provider before the credentials are accepted.
|
|
15
|
+
|
|
16
|
+
### Fixed
|
|
17
|
+
|
|
18
|
+
- Fixed DeepSeek vision models from losing image input while keeping image parts stripped for text-only DeepSeek endpoints.
|
|
19
|
+
- Fixed OpenAI-compatible gateways that report uppercase completion reasons such as `STOP` or `MAX_TOKENS`; these are now classified correctly, including mapping `MAX_TOKENS` to a length limit.
|
|
20
|
+
- Fixed provider message-count limit errors being treated as unrecoverable payload errors instead of recoverable context overflows.
|
|
21
|
+
- Improved Codex WebSocket continuations so rate limits, throttling, and compatible mode changes preserve valid response continuations instead of unnecessarily replaying the full context.
|
|
22
|
+
- Fixed Codex WebSocket cleanup failures caused by already-closed sockets.
|
|
23
|
+
- Added safe retries for transient mid-stream socket closures across OpenAI Responses, Chat Completions, Azure OpenAI Responses, and Codex SSE when no replay-unsafe output has been emitted.
|
|
24
|
+
- Fixed usage and cost reporting for OpenAI-compatible gateways backed by Vertex AI or Gemini by recognizing cached prompt tokens reported through `cachedContentTokenCount`.
|
|
25
|
+
|
|
26
|
+
## [18.0.4] - 2026-08-24
|
|
27
|
+
|
|
28
|
+
### Fixed
|
|
29
|
+
|
|
30
|
+
- Fixed Cursor tool calls through OpenAI-compatible authentication gateways losing arguments when complete argument maps are sent without streaming deltas ([#9479](https://github.com/can1357/oh-my-pi/issues/9479)).
|
|
31
|
+
- Fixed Cursor plan entitlement refusals repeatedly selecting ineligible accounts by scoping credential blocks to the requested model during rotation ([#9488](https://github.com/can1357/oh-my-pi/issues/9488)).
|
|
32
|
+
- Improved HTTP 413 error classification to accurately distinguish between payload/media size limits and token context window overflows, preventing inappropriate token compaction attempts and routing to correct recovery/fallback strategies ([#9235](https://github.com/can1357/oh-my-pi/issues/9235)).
|
|
33
|
+
- Fixed Cursor conversation rotation after aborts or mid-turn restarts to properly replay the last user message on a fresh conversation.
|
|
34
|
+
|
|
5
35
|
## [18.0.3] - 2026-08-23
|
|
6
36
|
|
|
7
37
|
### Fixed
|
package/README.md
CHANGED
|
@@ -60,6 +60,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
|
|
|
60
60
|
- **NVIDIA** (requires `NVIDIA_API_KEY`)
|
|
61
61
|
- **NanoGPT** (requires `NANO_GPT_API_KEY`)
|
|
62
62
|
- **Novita** (requires `NOVITA_API_KEY`)
|
|
63
|
+
- **DeepInfra** (requires `DEEPINFRA_API_KEY`)
|
|
63
64
|
- **Hugging Face Inference**
|
|
64
65
|
- **xAI**
|
|
65
66
|
- **Venice** (requires `VENICE_API_KEY`)
|
|
@@ -943,6 +944,7 @@ In Node.js environments, you can set environment variables to avoid passing API
|
|
|
943
944
|
| NVIDIA | `NVIDIA_API_KEY` |
|
|
944
945
|
| NanoGPT | `NANO_GPT_API_KEY` |
|
|
945
946
|
| Novita | `NOVITA_API_KEY` |
|
|
947
|
+
| DeepInfra | `DEEPINFRA_API_KEY` |
|
|
946
948
|
| Venice | `VENICE_API_KEY` |
|
|
947
949
|
| Moonshot | `MOONSHOT_API_KEY` |
|
|
948
950
|
| xAI | `XAI_API_KEY` |
|
|
@@ -983,6 +985,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
|
|
|
983
985
|
- NVIDIA: `https://integrate.api.nvidia.com/v1`
|
|
984
986
|
- NanoGPT: `https://nano-gpt.com/api/v1`
|
|
985
987
|
- Novita: `https://api.novita.ai/openai/v1`
|
|
988
|
+
- DeepInfra: `https://api.deepinfra.com/v1/openai`
|
|
986
989
|
- Hugging Face Inference: `https://router.huggingface.co/v1`
|
|
987
990
|
- Venice: `https://api.venice.ai/api/v1`
|
|
988
991
|
- Xiaomi MiMo: `https://api.xiaomimimo.com/anthropic`
|
|
@@ -1084,7 +1087,7 @@ Credentials are saved to `agent.db` in the agent directory. `/login qianfan` ope
|
|
|
1084
1087
|
|
|
1085
1088
|
`login` supports OAuth providers (Anthropic, OpenAI Codex, GitHub Copilot, Gemini CLI, Antigravity) and API-key onboarding flows.
|
|
1086
1089
|
|
|
1087
|
-
For the current API-key onboarding flows, the library covers Together, Moonshot, Qianfan, NVIDIA, NanoGPT, Novita, Hugging Face, Venice, Xiaomi, vLLM, LiteLLM, Cloudflare AI Gateway, Qwen Portal, and Ollama Cloud. Ollama remains the local runtime integration; set `OLLAMA_API_KEY` only when your local or self-hosted deployment enforces bearer auth.
|
|
1090
|
+
For the current API-key onboarding flows, the library covers Together, Moonshot, Qianfan, NVIDIA, NanoGPT, Novita, DeepInfra, Hugging Face, Venice, Xiaomi, vLLM, LiteLLM, Cloudflare AI Gateway, Qwen Portal, and Ollama Cloud. Ollama remains the local runtime integration; set `OLLAMA_API_KEY` only when your local or self-hosted deployment enforces bearer auth.
|
|
1088
1091
|
|
|
1089
1092
|
### Programmatic OAuth
|
|
1090
1093
|
|
|
@@ -1151,8 +1151,8 @@ export declare class AuthStorage {
|
|
|
1151
1151
|
* - usage-limit / account-rate-limit error → {@link AuthStorage.markUsageLimitReached}
|
|
1152
1152
|
* (temporary block via its own backoff — default plus server usage-report
|
|
1153
1153
|
* reset; sticky left intact so the next resolve re-ranks around the block).
|
|
1154
|
-
* - exact
|
|
1155
|
-
*
|
|
1154
|
+
* - exact model-entitlement denial (Codex ChatGPT account or Cursor plan) →
|
|
1155
|
+
* temporarily block only that requested model, then rotate.
|
|
1156
1156
|
* - other account-scoped policy denial → temporarily block that account
|
|
1157
1157
|
* without marking its credential suspect, then rotate through siblings.
|
|
1158
1158
|
* - otherwise (hard 401 / auth failure) → mark the credential suspect (or
|
|
@@ -23,6 +23,8 @@ export declare const Flag: {
|
|
|
23
23
|
readonly FastModeUnsupported: 536870912;
|
|
24
24
|
/** OAuth refresh failed definitively — the stored grant is dead, re-login required. */
|
|
25
25
|
readonly OAuthExpiry: 1073741824;
|
|
26
|
+
/** HTTP 413 byte/media rejection — token compaction cannot shrink bytes or media budgets (#9235). */
|
|
27
|
+
readonly PayloadRejected: 2147483648;
|
|
26
28
|
};
|
|
27
29
|
export type Flag = (typeof Flag)[keyof typeof Flag];
|
|
28
30
|
export declare const STREAM_READ_ERROR_PATTERN: RegExp;
|
|
@@ -65,6 +67,8 @@ export declare function isAccountPolicyError(error: unknown, api?: Api): boolean
|
|
|
65
67
|
export declare function codexChatGPTAccountPolicyModel(error: unknown, depth?: number): string | undefined;
|
|
66
68
|
/** Whether the exact Codex entitlement denial applies to this provider and requested model. */
|
|
67
69
|
export declare function isCodexChatGPTAccountPolicyError(error: unknown, provider: string, modelId: string | undefined): boolean;
|
|
70
|
+
/** Whether Cursor returned a non-retryable plan entitlement denial for this account. */
|
|
71
|
+
export declare function isCursorPlanAccountPolicyError(error: unknown, provider: string, depth?: number): boolean;
|
|
68
72
|
/**
|
|
69
73
|
* Strict-tool rejection: grammar too large, schema too complex, or structured
|
|
70
74
|
* outputs unsupported by the model/endpoint.
|
|
@@ -93,7 +97,16 @@ export declare function classifyMessage(message: {
|
|
|
93
97
|
errorStatus?: number;
|
|
94
98
|
}): number;
|
|
95
99
|
export declare function attach<E extends object>(error: E, id: number): E;
|
|
100
|
+
/** Provider-reported usage proves context-window excess — authoritative, compaction-owned (#9235). */
|
|
101
|
+
export declare function isUsageBackedContextOverflow(message: AssistantMessage, contextWindow?: number): boolean;
|
|
96
102
|
export declare function isContextOverflow(message: AssistantMessage, contextWindow?: number): boolean;
|
|
103
|
+
/** HTTP 413 byte/media rejection (#9235); may co-occur with {@link isContextOverflow} for bare `413 (no body)`.
|
|
104
|
+
* Callers with local headroom should skip compaction when this returns true. */
|
|
105
|
+
export declare function isPayloadRejection(message: AssistantMessage): boolean;
|
|
106
|
+
/** Dual-flagged 413 (PayloadRejected + ContextOverflow) with no provider-reported token excess (#9235).
|
|
107
|
+
* The co-flag means a different provider's larger byte/media budget may accept the request.
|
|
108
|
+
* Usage-backed overflows are authoritative window excesses and never ambiguous. */
|
|
109
|
+
export declare function isTextAmbiguousContextOverflow(errorId: number, message: AssistantMessage | undefined, contextWindow?: number): boolean;
|
|
97
110
|
export declare function stringify(id: number | undefined): string;
|
|
98
111
|
/**
|
|
99
112
|
* Transient stream corruption where the response was truncated mid-JSON.
|
|
@@ -218,11 +218,8 @@ export declare function isInvalidThinkingSignatureError(message: string): boolea
|
|
|
218
218
|
*/
|
|
219
219
|
export declare function maybeAddReplayUnsignedThinkingHint(model: Model<"anthropic-messages">, message: string): string;
|
|
220
220
|
/**
|
|
221
|
-
* Public entry:
|
|
222
|
-
*
|
|
223
|
-
* stall the agent loop). The inner attempt keeps its own provider-failure retry
|
|
224
|
-
* loop; this layer only re-issues a fresh request on an empty success. Shared
|
|
225
|
-
* with the OpenAI-completions provider via `withEmptyCompletionRetry`.
|
|
221
|
+
* Public entry: retry benign empty completions before they reach the agent
|
|
222
|
+
* loop. The inner attempt owns Anthropic provider-failure retries.
|
|
226
223
|
*/
|
|
227
224
|
export declare const streamAnthropic: StreamFunction<"anthropic-messages">;
|
|
228
225
|
export type AnthropicSystemBlock = {
|
|
@@ -12,6 +12,9 @@ export interface AzureOpenAIResponsesOptions extends StreamOptions {
|
|
|
12
12
|
disableReasoning?: boolean;
|
|
13
13
|
}
|
|
14
14
|
/**
|
|
15
|
-
*
|
|
15
|
+
* Retries transient Azure stream failures only before assistant output commits
|
|
16
|
+
* the attempt. The unsupported explicit prompt-cache config is rejected
|
|
17
|
+
* synchronously here — callers of the direct entrypoint get the immediate
|
|
18
|
+
* `ConfigurationError` rather than a stream whose `.result()` rejects later.
|
|
16
19
|
*/
|
|
17
20
|
export declare const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses">;
|
|
@@ -38,10 +38,8 @@ export interface OpenAICompletionsOptions extends StreamOptions {
|
|
|
38
38
|
promptCache?: OpenAIPromptCacheOptions;
|
|
39
39
|
}
|
|
40
40
|
/**
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
* "stop"` and no usage, which would otherwise stall the agent loop. Shared with
|
|
44
|
-
* the Anthropic provider via `withEmptyCompletionRetry`.
|
|
41
|
+
* Retries benign empty completions and transient provider failures only before
|
|
42
|
+
* assistant output commits the attempt.
|
|
45
43
|
*/
|
|
46
44
|
export declare const streamOpenAICompletions: StreamFunction<"openai-completions">;
|
|
47
45
|
export declare function parseChunkUsage(rawUsage: object, model: Model<"openai-completions">, premiumRequests: number | undefined): AssistantMessage["usage"];
|
|
@@ -106,10 +106,9 @@ type OpenAIResponsesSamplingParams = ResponseCreateParamsStreaming & {
|
|
|
106
106
|
cache_ttl?: "5m" | "1h";
|
|
107
107
|
};
|
|
108
108
|
/**
|
|
109
|
-
* Public entry:
|
|
110
|
-
*
|
|
111
|
-
*
|
|
112
|
-
* Anthropic providers via `withEmptyCompletionRetry`.
|
|
109
|
+
* Public entry: retry benign empty completions before they reach the agent
|
|
110
|
+
* loop. Transient stream failures are retried inside the attempt so stateful
|
|
111
|
+
* Responses request metadata remains stable.
|
|
113
112
|
*/
|
|
114
113
|
export declare const streamOpenAIResponses: StreamFunction<"openai-responses">;
|
|
115
114
|
export declare function buildParams(model: Model<"openai-responses">, context: Context, options: OpenAIResponsesOptions | undefined, providerSessionState: OpenAIResponsesProviderSessionState | undefined, strictToolsScope?: OpenAIStrictToolsScope, disableStrictToolsOverride?: boolean, statefulCacheBaseline?: ResponseInput): {
|
|
@@ -234,6 +234,7 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
|
|
|
234
234
|
preserve_thinking?: boolean;
|
|
235
235
|
chat_template_kwargs?: {
|
|
236
236
|
enable_thinking?: boolean;
|
|
237
|
+
thinking?: boolean;
|
|
237
238
|
preserve_thinking?: boolean;
|
|
238
239
|
reasoning_effort?: string;
|
|
239
240
|
};
|
|
@@ -684,4 +685,4 @@ export declare function buildResponsesDeltaInput<TItem extends ResponseInputItem
|
|
|
684
685
|
input?: TItem[];
|
|
685
686
|
} | undefined, previousResponseItems: readonly TItem[] | undefined, current: {
|
|
686
687
|
input?: TItem[];
|
|
687
|
-
}): TItem[] | null;
|
|
688
|
+
}, additionalTopLevelExcludeMap?: Readonly<Record<string, boolean>>): TItem[] | null;
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
export declare const loginDeepinfra: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
2
|
+
export declare const deepinfraProvider: {
|
|
3
|
+
id: "deepinfra";
|
|
4
|
+
name: string;
|
|
5
|
+
login: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
6
|
+
};
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Sign in with OpenRouter (OAuth PKCE).
|
|
3
|
+
*
|
|
4
|
+
* No client registration: the S256 PKCE challenge is the only proof of
|
|
5
|
+
* identity. The browser authorizes at `https://openrouter.ai/auth`, redirects
|
|
6
|
+
* back to the loopback callback with `?code=`, and the code + verifier
|
|
7
|
+
* exchange at `/api/v1/auth/keys` mints a durable `sk-or-…` API key.
|
|
8
|
+
*
|
|
9
|
+
* OpenRouter never echoes a `state` parameter (the redirect only appends
|
|
10
|
+
* `?code=` to the callback URL), so the flow runs with an empty expected
|
|
11
|
+
* state — callback-state validation is disabled and the PKCE verifier binds
|
|
12
|
+
* the exchange instead.
|
|
13
|
+
*
|
|
14
|
+
* The manual-input race doubles as the API-key paste path: pasted `sk-or-…`
|
|
15
|
+
* keys skip the PKCE exchange and are validated against `/api/v1/auth/key`
|
|
16
|
+
* (the canonical "who am I" endpoint — `/api/v1/models` returns 200 for any
|
|
17
|
+
* bearer), so one login entry covers both browser sign-in and key paste.
|
|
18
|
+
*/
|
|
19
|
+
import type { FetchImpl } from "../../types.js";
|
|
20
|
+
import { OAuthCallbackFlow } from "./callback-server.js";
|
|
21
|
+
import type { OAuthController, OAuthCredentials } from "./types.js";
|
|
22
|
+
/** Exchange an authorization code + PKCE verifier for a durable OpenRouter API key. */
|
|
23
|
+
export declare function exchangeOpenRouterCode(code: string, codeVerifier: string, fetchImpl?: FetchImpl): Promise<string>;
|
|
24
|
+
export declare class OpenRouterOAuthFlow extends OAuthCallbackFlow {
|
|
25
|
+
#private;
|
|
26
|
+
constructor(ctrl: OAuthController);
|
|
27
|
+
/** OpenRouter never echoes `state`; empty state disables callback-state validation. */
|
|
28
|
+
generateState(): string;
|
|
29
|
+
generateAuthUrl(_state: string, redirectUri: string): Promise<{
|
|
30
|
+
url: string;
|
|
31
|
+
instructions?: string;
|
|
32
|
+
}>;
|
|
33
|
+
exchangeToken(code: string): Promise<OAuthCredentials>;
|
|
34
|
+
}
|
|
35
|
+
/** Log in with Sign in with OpenRouter (PKCE); mints a durable API key. */
|
|
36
|
+
export declare function loginOpenRouterOAuth(ctrl: OAuthController): Promise<OAuthCredentials>;
|
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
import type { OAuthLoginCallbacks } from "./oauth/types.js";
|
|
2
|
-
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
2
|
+
/**
|
|
3
|
+
* OpenRouter login: Sign in with OpenRouter (OAuth PKCE) that mints a durable
|
|
4
|
+
* `sk-or-…` API key, with the manual-input race accepting a pasted existing
|
|
5
|
+
* key (validated via `/api/v1/auth/key`). Either path resolves to a plain API
|
|
6
|
+
* key string, stored as an `api_key` credential.
|
|
7
7
|
*/
|
|
8
|
-
export declare const loginOpenRouter: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
9
8
|
export declare const openrouterProvider: {
|
|
10
9
|
readonly id: "openrouter";
|
|
11
10
|
readonly name: "OpenRouter";
|
|
12
11
|
readonly login: (cb: OAuthLoginCallbacks) => Promise<string>;
|
|
12
|
+
readonly callbackPort: 54549;
|
|
13
|
+
readonly pasteCodeFlow: true;
|
|
13
14
|
};
|
|
@@ -82,6 +82,10 @@ declare const ALL: ({
|
|
|
82
82
|
readonly name: "Cursor (Claude, GPT, etc.)";
|
|
83
83
|
readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<import("./oauth/index.js").OAuthCredentials>;
|
|
84
84
|
readonly refreshToken: (credentials: import("./oauth/index.js").OAuthCredentials) => Promise<import("./oauth/index.js").OAuthCredentials>;
|
|
85
|
+
} | {
|
|
86
|
+
id: "deepinfra";
|
|
87
|
+
name: string;
|
|
88
|
+
login: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
85
89
|
} | {
|
|
86
90
|
readonly id: "deepseek";
|
|
87
91
|
readonly name: "DeepSeek";
|
|
@@ -254,6 +258,8 @@ declare const ALL: ({
|
|
|
254
258
|
readonly id: "openrouter";
|
|
255
259
|
readonly name: "OpenRouter";
|
|
256
260
|
readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string>;
|
|
261
|
+
readonly callbackPort: 54549;
|
|
262
|
+
readonly pasteCodeFlow: true;
|
|
257
263
|
} | {
|
|
258
264
|
readonly id: "parallel";
|
|
259
265
|
readonly name: "Parallel";
|
|
@@ -342,6 +348,10 @@ declare const ALL: ({
|
|
|
342
348
|
readonly id: "xiaomi-token-plan-sgp";
|
|
343
349
|
readonly name: "Xiaomi Token Plan (Singapore)";
|
|
344
350
|
readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string>;
|
|
351
|
+
} | {
|
|
352
|
+
readonly id: "yolo-auto";
|
|
353
|
+
readonly name: "Yolo-Auto";
|
|
354
|
+
readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string>;
|
|
345
355
|
} | {
|
|
346
356
|
readonly id: "zai";
|
|
347
357
|
readonly name: "Z.AI (GLM Coding Plan)";
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import type { OAuthLoginCallbacks } from "./oauth/types.js";
|
|
2
|
+
/**
|
|
3
|
+
* Yolo-Auto login flow (API key paste, validated via `/v1/models`).
|
|
4
|
+
*
|
|
5
|
+
* Yolo-Auto is a flat-rate OpenAI-compatible API. `GET /v1/models` 401s any
|
|
6
|
+
* missing or invalid bearer (verified against the live endpoint), so it doubles
|
|
7
|
+
* as the canonical "who am I" check — the same role OpenRouter's `/auth/key`
|
|
8
|
+
* plays there.
|
|
9
|
+
*/
|
|
10
|
+
export declare const loginYoloAuto: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
11
|
+
export declare const yoloAutoProvider: {
|
|
12
|
+
readonly id: "yolo-auto";
|
|
13
|
+
readonly name: "Yolo-Auto";
|
|
14
|
+
readonly login: (cb: OAuthLoginCallbacks) => Promise<string>;
|
|
15
|
+
};
|
|
@@ -8,15 +8,24 @@ export declare const EMPTY_COMPLETION_BASE_DELAY_MS = 500;
|
|
|
8
8
|
* — or one that only ever produced thinking — is the "empty response" failure.
|
|
9
9
|
*/
|
|
10
10
|
export declare function hasVisibleAssistantContent(message: AssistantMessage): boolean;
|
|
11
|
-
interface
|
|
11
|
+
interface StreamRetryOptions {
|
|
12
12
|
signal?: AbortSignal;
|
|
13
13
|
providerRetryWait?: (delayMs: number, signal?: AbortSignal) => Promise<void>;
|
|
14
14
|
acceptEmptyResponse?: boolean;
|
|
15
15
|
}
|
|
16
|
+
/** Controls which replay-safe provider results may issue a fresh request. */
|
|
17
|
+
export interface ReplaySafeStreamRetryPolicy {
|
|
18
|
+
/** Retry benign terminal stops that contain no visible output. */
|
|
19
|
+
retryEmptyCompletion?: boolean;
|
|
20
|
+
/** Retry transient provider errors before output is committed. */
|
|
21
|
+
retryProviderErrors?: boolean;
|
|
22
|
+
/** Maximum transient provider-error retries; empty completions keep their shared fixed budget. */
|
|
23
|
+
maxProviderErrorRetries?: number;
|
|
24
|
+
}
|
|
16
25
|
/**
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
26
|
+
* Re-issues a fresh provider request only while the current attempt remains
|
|
27
|
+
* replay-safe. Buffered pre-output events from discarded attempts never reach
|
|
28
|
+
* consumers.
|
|
20
29
|
*/
|
|
21
|
-
export declare function
|
|
30
|
+
export declare function withReplaySafeStreamRetry<M, O extends StreamRetryOptions>(model: M, context: Context, options: O | undefined, attempt: (model: M, context: Context, options?: O) => AssistantMessageEventStream, policy: ReplaySafeStreamRetryPolicy): AssistantMessageEventStream;
|
|
22
31
|
export {};
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-ai",
|
|
4
|
-
"version": "18.0.
|
|
4
|
+
"version": "18.0.5",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": "Stencil Labs, Inc.",
|
|
@@ -37,10 +37,10 @@
|
|
|
37
37
|
"fmt": "biome format --write ."
|
|
38
38
|
},
|
|
39
39
|
"dependencies": {
|
|
40
|
-
"@oh-my-pi/omptype": "18.0.
|
|
41
|
-
"@oh-my-pi/pi-catalog": "18.0.
|
|
42
|
-
"@oh-my-pi/pi-utils": "18.0.
|
|
43
|
-
"@oh-my-pi/pi-wire": "18.0.
|
|
40
|
+
"@oh-my-pi/omptype": "18.0.5",
|
|
41
|
+
"@oh-my-pi/pi-catalog": "18.0.5",
|
|
42
|
+
"@oh-my-pi/pi-utils": "18.0.5",
|
|
43
|
+
"@oh-my-pi/pi-wire": "18.0.5"
|
|
44
44
|
},
|
|
45
45
|
"devDependencies": {
|
|
46
46
|
"@types/bun": "^1.3.14"
|
package/src/auth-storage.ts
CHANGED
|
@@ -1028,9 +1028,13 @@ function resolveOpenAICodexPlanRequirement(provider: string, modelId: string | u
|
|
|
1028
1028
|
}
|
|
1029
1029
|
|
|
1030
1030
|
const MODEL_ACCOUNT_POLICY_BLOCK_SCOPE_PREFIX = "model-policy:";
|
|
1031
|
+
const MODEL_ACCOUNT_POLICY_PROVIDERS: Readonly<Record<string, true>> = {
|
|
1032
|
+
"openai-codex": true,
|
|
1033
|
+
cursor: true,
|
|
1034
|
+
};
|
|
1031
1035
|
|
|
1032
1036
|
function modelAccountPolicyBlockScope(provider: string, modelId: string | undefined): string | undefined {
|
|
1033
|
-
if (provider
|
|
1037
|
+
if (!Object.hasOwn(MODEL_ACCOUNT_POLICY_PROVIDERS, provider) || typeof modelId !== "string") return undefined;
|
|
1034
1038
|
const separator = modelId.lastIndexOf("/");
|
|
1035
1039
|
const bareModelId = (separator === -1 ? modelId : modelId.slice(separator + 1)).trim().toLowerCase();
|
|
1036
1040
|
if (!bareModelId || bareModelId.includes("\0")) return undefined;
|
|
@@ -6393,8 +6397,8 @@ export class AuthStorage {
|
|
|
6393
6397
|
* - usage-limit / account-rate-limit error → {@link AuthStorage.markUsageLimitReached}
|
|
6394
6398
|
* (temporary block via its own backoff — default plus server usage-report
|
|
6395
6399
|
* reset; sticky left intact so the next resolve re-ranks around the block).
|
|
6396
|
-
* - exact
|
|
6397
|
-
*
|
|
6400
|
+
* - exact model-entitlement denial (Codex ChatGPT account or Cursor plan) →
|
|
6401
|
+
* temporarily block only that requested model, then rotate.
|
|
6398
6402
|
* - other account-scoped policy denial → temporarily block that account
|
|
6399
6403
|
* without marking its credential suspect, then rotate through siblings.
|
|
6400
6404
|
* - otherwise (hard 401 / auth failure) → mark the credential suspect (or
|
|
@@ -6411,7 +6415,9 @@ export class AuthStorage {
|
|
|
6411
6415
|
const error = options?.error;
|
|
6412
6416
|
const status = AIError.status(error);
|
|
6413
6417
|
const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined;
|
|
6414
|
-
|
|
6418
|
+
const exactCursorModelPolicy = AIError.isCursorPlanAccountPolicyError(error, provider);
|
|
6419
|
+
const accountPolicy = exactCursorModelPolicy || AIError.isAccountPolicyError(error);
|
|
6420
|
+
if (!accountPolicy && (AIError.isUsageLimit(error) || isUsageLimitOutcome(status, message))) {
|
|
6415
6421
|
// Thread the provider-specified reset window (e.g. Devin "Your limit
|
|
6416
6422
|
// will reset in 13 minutes") into the block duration so the credential
|
|
6417
6423
|
// is not reselected and hammered while the cap remains active.
|
|
@@ -6436,15 +6442,16 @@ export class AuthStorage {
|
|
|
6436
6442
|
const deniedModel = AIError.codexChatGPTAccountPolicyModel(error);
|
|
6437
6443
|
const exactCodexModelPolicy =
|
|
6438
6444
|
deniedModel !== undefined && AIError.isCodexChatGPTAccountPolicyError(error, provider, options?.modelId);
|
|
6445
|
+
const exactModelPolicy = exactCodexModelPolicy || exactCursorModelPolicy;
|
|
6439
6446
|
// The exact sentence is provider-controlled input. A non-Codex provider,
|
|
6440
6447
|
// absent request model, or mismatched model must not turn it into either a
|
|
6441
6448
|
// global block or a hard-auth invalidation.
|
|
6442
6449
|
if (deniedModel !== undefined && !exactCodexModelPolicy) return false;
|
|
6443
|
-
if (
|
|
6444
|
-
const modelPolicyScope =
|
|
6450
|
+
if (exactModelPolicy || accountPolicy) {
|
|
6451
|
+
const modelPolicyScope = exactModelPolicy
|
|
6445
6452
|
? modelAccountPolicyBlockScope(provider, options?.modelId)
|
|
6446
6453
|
: undefined;
|
|
6447
|
-
if (
|
|
6454
|
+
if (exactModelPolicy && modelPolicyScope === undefined) return false;
|
|
6448
6455
|
const routing = this.#credentialBlockRouting(
|
|
6449
6456
|
provider,
|
|
6450
6457
|
sessionCredential.type,
|