@oh-my-pi/pi-ai 18.0.7 → 18.0.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/README.md +4 -5
- package/dist/types/auth-storage.d.ts +12 -0
- package/dist/types/registry/cloudflare-ai-gateway.d.ts +8 -8
- package/dist/types/registry/registry.d.ts +5 -0
- package/dist/types/registry/types.d.ts +3 -0
- package/dist/types/types.d.ts +6 -0
- package/dist/types/usage/google-antigravity.d.ts +12 -1
- package/dist/types/usage.d.ts +14 -10
- package/package.json +5 -5
- package/src/auth/sqlite-credential-store.ts +12 -4
- package/src/auth-broker/wire-schemas.ts +1 -1
- package/src/auth-storage.ts +40 -4
- package/src/providers/amazon-bedrock.ts +9 -0
- package/src/providers/cursor.ts +38 -23
- package/src/registry/cloudflare-ai-gateway.ts +120 -16
- package/src/registry/oauth/oauth.html +8 -2
- package/src/registry/types.ts +3 -0
- package/src/stream.ts +9 -7
- package/src/types.ts +6 -0
- package/src/usage/google-antigravity.ts +25 -13
- package/src/usage/openai-codex.ts +5 -2
- package/src/usage/zai.ts +50 -4
- package/src/usage.ts +14 -4
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,25 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.0.9] - 2026-08-28
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Improved OAuth sign-in flows, including a fallback message when the browser cannot automatically close the OAuth success tab.
|
|
10
|
+
- Fixed Cloudflare AI Gateway onboarding and routing so gateway account and endpoint configuration is preserved correctly while gateway credentials are not sent as upstream OpenAI authorization headers.
|
|
11
|
+
- Fixed Codex OAuth quota handling so chat and Spark usage remain independent, legacy shared quota limits continue to work, and incomplete usage reports are not incorrectly treated as unlimited.
|
|
12
|
+
|
|
13
|
+
## [18.0.8] - 2026-08-27
|
|
14
|
+
|
|
15
|
+
### Added
|
|
16
|
+
|
|
17
|
+
- Added Z.AI GLM Coding Plan usage tracking: credit-based `CREDIT_LIMIT` windows (5h + weekly) now surface in `omp usage` and the status line with the plan tier (`plan: lite/pro/max`).
|
|
18
|
+
|
|
19
|
+
### Fixed
|
|
20
|
+
|
|
21
|
+
- Fixed Amazon Bedrock requests to OpenAI-schema models (the `gpt-5.x` SKUs) failing with HTTP 400 `unknown_parameter: 'thinking'` when reasoning was enabled, by sending `reasoning.effort` instead of Anthropic's `thinking` budget block for models the catalog marks as effort-controlled.
|
|
22
|
+
- Fixed Cursor replay rejecting sessions with orphaned tool results while preserving their output as assistant context.
|
|
23
|
+
|
|
5
24
|
## [18.0.7] - 2026-08-26
|
|
6
25
|
|
|
7
26
|
### Added
|
package/README.md
CHANGED
|
@@ -76,7 +76,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
|
|
|
76
76
|
- **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
|
|
77
77
|
- **QwenCloud Token Plan** (supports `/login alibaba-token-plan`, `ALIBABA_TOKEN_PLAN_API_KEY`, or `BAILIAN_TOKEN_PLAN_API_KEY`; interactive login first selects a region — International (Singapore, default), China (Beijing) for 百炼 Token Plan keys, or a custom base URL — since region keys are non-interchangeable, then optionally stores a `home.qwencloud.com` Cookie request header for best-effort 5-hour and 7-day quota reporting)
|
|
78
78
|
To enable quota reporting, sign in to the Token Plan dashboard, copy the `Cookie` request-header value from a `home.qwencloud.com` request in browser developer tools, and paste it at the second login prompt. Press Enter to skip; the Cookie is sensitive and session-lived, so rerun login when it expires.
|
|
79
|
-
- **Cloudflare AI Gateway** (
|
|
79
|
+
- **Cloudflare AI Gateway** (supports `/login cloudflare-ai-gateway`, or `CLOUDFLARE_AI_GATEWAY_API_KEY` with `CLOUDFLARE_ACCOUNT_ID` and `CLOUDFLARE_GATEWAY_ID`)
|
|
80
80
|
- **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
|
|
81
81
|
- **Ollama Cloud** (hosted native Ollama API; requires `OLLAMA_CLOUD_API_KEY`)
|
|
82
82
|
- **llama.cpp** (local OpenAI and Anthropic compatible inference server)
|
|
@@ -960,11 +960,10 @@ In Node.js environments, you can set environment variables to avoid passing API
|
|
|
960
960
|
| Xiaomi MiMo | `XIAOMI_API_KEY` |
|
|
961
961
|
| ZenMux | `ZENMUX_API_KEY` |
|
|
962
962
|
| vLLM | `VLLM_API_KEY` |
|
|
963
|
-
| Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY`
|
|
963
|
+
| Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` + `CLOUDFLARE_ACCOUNT_ID` + `CLOUDFLARE_GATEWAY_ID` |
|
|
964
964
|
| GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
|
|
965
965
|
|
|
966
|
-
For Cloudflare
|
|
967
|
-
`https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic`.
|
|
966
|
+
`/login cloudflare-ai-gateway` collects and stores the gateway token, account ID, and gateway ID. For environment configuration, set all three Cloudflare values above. OMP derives provider endpoints from the account and gateway IDs.
|
|
968
967
|
|
|
969
968
|
For Anthropic Foundry routing, set `CLAUDE_CODE_USE_FOUNDRY=true` plus:
|
|
970
969
|
`FOUNDRY_BASE_URL`, `ANTHROPIC_FOUNDRY_API_KEY`, optional `ANTHROPIC_CUSTOM_HEADERS`,
|
|
@@ -996,7 +995,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
|
|
|
996
995
|
- Ollama: local OpenAI-compatible runtime (`http://127.0.0.1:11434/v1`)
|
|
997
996
|
- Ollama Cloud: native Ollama API host (`https://ollama.com/api`, configured here as base URL `https://ollama.com`)
|
|
998
997
|
- LiteLLM: `http://localhost:4000/v1`
|
|
999
|
-
- Cloudflare AI Gateway: `https://gateway.ai.cloudflare.com/v1/<account>/<gateway
|
|
998
|
+
- Cloudflare AI Gateway: native Anthropic, OpenAI, and Workers AI routes under `https://gateway.ai.cloudflare.com/v1/<account>/<gateway>`
|
|
1000
999
|
- Qwen Portal: `https://portal.qwen.ai/v1`
|
|
1001
1000
|
When set, the library automatically uses these keys:
|
|
1002
1001
|
|
|
@@ -821,6 +821,18 @@ export declare class AuthStorage {
|
|
|
821
821
|
* `XAI_API_KEY` does not auto-select SuperGrok (`xai-oauth`).
|
|
822
822
|
*/
|
|
823
823
|
hasAuth(provider: string): boolean;
|
|
824
|
+
/**
|
|
825
|
+
* Like {@link hasAuth} but excludes providers whose only credential is the
|
|
826
|
+
* self-resolving {@link AUTHENTICATED_SENTINEL} — the marker AWS/Vertex
|
|
827
|
+
* transports return when a credential *source* merely exists (a stray
|
|
828
|
+
* `~/.aws` profile, an EC2 instance role, Application Default Credentials)
|
|
829
|
+
* without a usable key resolved yet. Default-model auto-selection uses this
|
|
830
|
+
* so an ambiently-available provider (e.g. `amazon-bedrock` via an unrelated
|
|
831
|
+
* AWS profile) does not win the startup default over a provider the user
|
|
832
|
+
* actually signed into and then 403 on the first turn. Explicit selection
|
|
833
|
+
* and picker visibility still go through {@link hasAuth}. See issue #9967.
|
|
834
|
+
*/
|
|
835
|
+
hasConcreteAuth(provider: string): boolean;
|
|
824
836
|
/**
|
|
825
837
|
* Whether a request could resolve a key for this provider, including
|
|
826
838
|
* cross-provider env aliases (`xai-oauth` borrowing `XAI_API_KEY`).
|
|
@@ -1,13 +1,13 @@
|
|
|
1
|
-
import type { OAuthLoginCallbacks } from "./oauth/types.js";
|
|
2
|
-
/**
|
|
3
|
-
|
|
4
|
-
*
|
|
5
|
-
* Opens browser to Cloudflare AI Gateway authentication docs and prompts for a gateway token/API key.
|
|
6
|
-
* Returns the API key directly (not OAuthCredentials - this isn't OAuth).
|
|
7
|
-
*/
|
|
8
|
-
export declare const loginCloudflareAiGateway: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
|
|
1
|
+
import type { OAuthController, OAuthLoginCallbacks } from "./oauth/types.js";
|
|
2
|
+
/** Collect the gateway credential used by CLI, setup-wizard, and TUI login callers. */
|
|
3
|
+
export declare function loginCloudflareAiGateway(options: OAuthController): Promise<string>;
|
|
9
4
|
export declare const cloudflareAiGatewayProvider: {
|
|
10
5
|
readonly id: "cloudflare-ai-gateway";
|
|
11
6
|
readonly name: "Cloudflare AI Gateway";
|
|
7
|
+
readonly prepareModel: (model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>) => import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>;
|
|
8
|
+
readonly prepareRequest: (model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>, options: import("../index.js").StreamOptions) => {
|
|
9
|
+
model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>;
|
|
10
|
+
options: import("../index.js").StreamOptions;
|
|
11
|
+
};
|
|
12
12
|
readonly login: (cb: OAuthLoginCallbacks) => Promise<string>;
|
|
13
13
|
};
|
|
@@ -72,6 +72,11 @@ declare const ALL: ({
|
|
|
72
72
|
} | {
|
|
73
73
|
readonly id: "cloudflare-ai-gateway";
|
|
74
74
|
readonly name: "Cloudflare AI Gateway";
|
|
75
|
+
readonly prepareModel: (model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>) => import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>;
|
|
76
|
+
readonly prepareRequest: (model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>, options: import("../index.js").StreamOptions) => {
|
|
77
|
+
model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>;
|
|
78
|
+
options: import("../index.js").StreamOptions;
|
|
79
|
+
};
|
|
75
80
|
readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string>;
|
|
76
81
|
} | {
|
|
77
82
|
readonly id: "coreweave";
|
|
@@ -24,6 +24,7 @@ export interface PreparedProviderRequest {
|
|
|
24
24
|
readonly options: StreamOptions;
|
|
25
25
|
}
|
|
26
26
|
export type ProviderRequestPreparer = (model: Model<Api>, options: StreamOptions) => PreparedProviderRequest;
|
|
27
|
+
export type ProviderModelPreparer = (model: Model<Api>) => Model<Api>;
|
|
27
28
|
export type ProviderSimpleOptionsMapper = (options: SimpleStreamOptions) => Readonly<Record<string, unknown>>;
|
|
28
29
|
export interface ProviderModelDiscoveryConfig {
|
|
29
30
|
readonly apiKey?: string;
|
|
@@ -57,6 +58,8 @@ export interface ProviderDefinition {
|
|
|
57
58
|
readonly envKeys?: KeyResolver;
|
|
58
59
|
/** Provider transport can authenticate without a resolved API-key string. */
|
|
59
60
|
readonly allowsMissingApiKey?: boolean;
|
|
61
|
+
/** Provider-owned model normalization that must run before API-specific option mapping. */
|
|
62
|
+
readonly prepareModel?: ProviderModelPreparer;
|
|
60
63
|
/** Provider-owned request shaping applied before generic API dispatch. */
|
|
61
64
|
readonly prepareRequest?: ProviderRequestPreparer;
|
|
62
65
|
/** Provider-owned projection from the generic simple-stream option bag. */
|
package/dist/types/types.d.ts
CHANGED
|
@@ -698,6 +698,10 @@ export interface DeveloperMessage {
|
|
|
698
698
|
content: string | (TextContent | ImageContent)[];
|
|
699
699
|
/** Who initiated this message for billing/attribution semantics. */
|
|
700
700
|
attribution?: MessageAttribution;
|
|
701
|
+
/** True if the message was injected by the system (e.g., auto-continue) and initiates a fresh run rather than continuing the current one. */
|
|
702
|
+
synthetic?: boolean;
|
|
703
|
+
/** True when the synthetic prompt was a deliberate operator action (`.`, `c` continue shortcut) rather than an automatic continuation — its timestamp is the turn's prompt time. */
|
|
704
|
+
userInitiated?: boolean;
|
|
701
705
|
/** Provider-specific opaque payload used to reconstruct transport-native history. */
|
|
702
706
|
providerPayload?: ProviderPayload;
|
|
703
707
|
timestamp: number;
|
|
@@ -781,6 +785,8 @@ export interface AssistantMessage {
|
|
|
781
785
|
timestamp: number;
|
|
782
786
|
duration?: number;
|
|
783
787
|
ttft?: number;
|
|
788
|
+
/** Local wall-clock time the response finished streaming (ms since epoch); stamped by the session at message_end so prompt→yield timing never depends on provider-reported duration. */
|
|
789
|
+
completedAt?: number;
|
|
784
790
|
}
|
|
785
791
|
export interface ToolResultMessage<TDetails = unknown> {
|
|
786
792
|
role: "toolResult";
|
|
@@ -1,5 +1,16 @@
|
|
|
1
|
-
import type { CredentialRankingStrategy, UsageProvider } from "../usage.js";
|
|
1
|
+
import type { CredentialRankingContext, CredentialRankingStrategy, UsageLimit, UsageProvider, UsageReport } from "../usage.js";
|
|
2
2
|
export declare const antigravityUsageProvider: UsageProvider;
|
|
3
|
+
/** Map an Antigravity model id to its backend quota-counter key. */
|
|
4
|
+
export declare function getAntigravityCounterKeyForModel(modelId: string | undefined): string | undefined;
|
|
5
|
+
/**
|
|
6
|
+
* Scope an Antigravity report to the active model's backend counter, falling
|
|
7
|
+
* back to legacy default counters only when that backend has no limits.
|
|
8
|
+
*
|
|
9
|
+
* Exhaustion checks are only safe with a concrete backend counter. A no-model
|
|
10
|
+
* credential lookup (for example image-provider discovery) must not turn one
|
|
11
|
+
* exhausted family into a provider-wide block.
|
|
12
|
+
*/
|
|
13
|
+
export declare function scopeAntigravityLimitsForModel(report: UsageReport, context: CredentialRankingContext | undefined): UsageLimit[];
|
|
3
14
|
/**
|
|
4
15
|
* Antigravity quotas are returned per backend counter (Anthropic / Google /
|
|
5
16
|
* OpenAI) and can include both daily and weekly windows. `fetchAntigravityUsage`
|
package/dist/types/usage.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { FetchImpl, Provider } from "./types.js";
|
|
2
|
-
export type UsageUnit = "percent" | "tokens" | "requests" | "usd" | "minutes" | "bytes" | "unknown";
|
|
2
|
+
export type UsageUnit = "percent" | "tokens" | "requests" | "credits" | "usd" | "minutes" | "bytes" | "unknown";
|
|
3
3
|
export type UsageStatus = "ok" | "warning" | "exhausted" | "unknown";
|
|
4
4
|
/** Time window for a limit (e.g. 5h, 7d, monthly). */
|
|
5
5
|
export interface UsageWindow {
|
|
@@ -204,7 +204,7 @@ export interface ClientUsageClientSummary {
|
|
|
204
204
|
export interface ClientUsageSummary {
|
|
205
205
|
clients: ClientUsageClientSummary[];
|
|
206
206
|
}
|
|
207
|
-
export declare const usageUnitSchema: import("@oh-my-pi/omptype").FluentType<"bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd", "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd">;
|
|
207
|
+
export declare const usageUnitSchema: import("@oh-my-pi/omptype").FluentType<"bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd", "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd">;
|
|
208
208
|
export declare const usageStatusSchema: import("@oh-my-pi/omptype").FluentType<"exhausted" | "ok" | "unknown" | "warning", "exhausted" | "ok" | "unknown" | "warning">;
|
|
209
209
|
export declare const usageWindowSchema: import("@oh-my-pi/omptype").FluentType<{
|
|
210
210
|
durationMs?: number | undefined;
|
|
@@ -223,14 +223,14 @@ export declare const usageAmountSchema: import("@oh-my-pi/omptype").FluentType<{
|
|
|
223
223
|
limit?: number | undefined;
|
|
224
224
|
remaining?: number | undefined;
|
|
225
225
|
remainingFraction?: number | undefined;
|
|
226
|
-
unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
226
|
+
unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
227
227
|
used?: number | undefined;
|
|
228
228
|
usedFraction?: number | undefined;
|
|
229
229
|
}, {
|
|
230
230
|
limit?: number | undefined;
|
|
231
231
|
remaining?: number | undefined;
|
|
232
232
|
remainingFraction?: number | undefined;
|
|
233
|
-
unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
233
|
+
unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
234
234
|
used?: number | undefined;
|
|
235
235
|
usedFraction?: number | undefined;
|
|
236
236
|
}>;
|
|
@@ -258,7 +258,7 @@ export declare const usageLimitSchema: import("@oh-my-pi/omptype").FluentType<{
|
|
|
258
258
|
limit?: number | undefined;
|
|
259
259
|
remaining?: number | undefined;
|
|
260
260
|
remainingFraction?: number | undefined;
|
|
261
|
-
unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
261
|
+
unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
262
262
|
used?: number | undefined;
|
|
263
263
|
usedFraction?: number | undefined;
|
|
264
264
|
};
|
|
@@ -288,7 +288,7 @@ export declare const usageLimitSchema: import("@oh-my-pi/omptype").FluentType<{
|
|
|
288
288
|
limit?: number | undefined;
|
|
289
289
|
remaining?: number | undefined;
|
|
290
290
|
remainingFraction?: number | undefined;
|
|
291
|
-
unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
291
|
+
unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
292
292
|
used?: number | undefined;
|
|
293
293
|
usedFraction?: number | undefined;
|
|
294
294
|
};
|
|
@@ -345,7 +345,7 @@ export declare const usageReportSchema: import("@oh-my-pi/omptype").FluentType<{
|
|
|
345
345
|
limit?: number | undefined;
|
|
346
346
|
remaining?: number | undefined;
|
|
347
347
|
remainingFraction?: number | undefined;
|
|
348
|
-
unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
348
|
+
unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
349
349
|
used?: number | undefined;
|
|
350
350
|
usedFraction?: number | undefined;
|
|
351
351
|
};
|
|
@@ -390,7 +390,7 @@ export declare const usageReportSchema: import("@oh-my-pi/omptype").FluentType<{
|
|
|
390
390
|
limit?: number | undefined;
|
|
391
391
|
remaining?: number | undefined;
|
|
392
392
|
remainingFraction?: number | undefined;
|
|
393
|
-
unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
393
|
+
unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
|
|
394
394
|
used?: number | undefined;
|
|
395
395
|
usedFraction?: number | undefined;
|
|
396
396
|
};
|
|
@@ -528,6 +528,10 @@ export interface CredentialRankingStrategy {
|
|
|
528
528
|
primaryMs: number;
|
|
529
529
|
secondaryMs: number;
|
|
530
530
|
};
|
|
531
|
-
/**
|
|
532
|
-
|
|
531
|
+
/**
|
|
532
|
+
* Optional: priority boost for specific credential states (e.g., fresh 5h
|
|
533
|
+
* ticker start). `primaryUncapped` is true only when the fetched report has
|
|
534
|
+
* an applicable secondary window but no applicable primary window.
|
|
535
|
+
*/
|
|
536
|
+
hasPriorityBoost?(primary: UsageLimit | undefined, primaryUncapped?: boolean, context?: CredentialRankingContext): boolean;
|
|
533
537
|
}
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-ai",
|
|
4
|
-
"version": "18.0.
|
|
4
|
+
"version": "18.0.9",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": "Stencil Labs, Inc.",
|
|
@@ -37,10 +37,10 @@
|
|
|
37
37
|
"fmt": "biome format --write ."
|
|
38
38
|
},
|
|
39
39
|
"dependencies": {
|
|
40
|
-
"@oh-my-pi/omptype": "18.0.
|
|
41
|
-
"@oh-my-pi/pi-catalog": "18.0.
|
|
42
|
-
"@oh-my-pi/pi-utils": "18.0.
|
|
43
|
-
"@oh-my-pi/pi-wire": "18.0.
|
|
40
|
+
"@oh-my-pi/omptype": "18.0.9",
|
|
41
|
+
"@oh-my-pi/pi-catalog": "18.0.9",
|
|
42
|
+
"@oh-my-pi/pi-utils": "18.0.9",
|
|
43
|
+
"@oh-my-pi/pi-wire": "18.0.9"
|
|
44
44
|
},
|
|
45
45
|
"devDependencies": {
|
|
46
46
|
"@types/bun": "^1.3.14"
|
|
@@ -8,6 +8,7 @@ import { Database, type Statement } from "bun:sqlite";
|
|
|
8
8
|
import * as fs from "node:fs/promises";
|
|
9
9
|
import * as path from "node:path";
|
|
10
10
|
import { parseAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
|
|
11
|
+
import { parseCloudflareAiGatewayCredential } from "@oh-my-pi/pi-catalog/wire/cloudflare-ai-gateway";
|
|
11
12
|
import {
|
|
12
13
|
getAgentDbPath,
|
|
13
14
|
getDbBusyTimeoutMs,
|
|
@@ -215,10 +216,17 @@ function matchesReplacementCredential(
|
|
|
215
216
|
if (incoming.type === "api_key") {
|
|
216
217
|
if (existing.type !== "api_key") return false;
|
|
217
218
|
if (existing.key === incoming.key) return true;
|
|
218
|
-
if (provider
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
219
|
+
if (provider === "alibaba-token-plan") {
|
|
220
|
+
const existingToken = parseAlibabaTokenPlanCredential(existing.key)?.token;
|
|
221
|
+
const incomingToken = parseAlibabaTokenPlanCredential(incoming.key)?.token;
|
|
222
|
+
return existingToken !== undefined && existingToken === incomingToken;
|
|
223
|
+
}
|
|
224
|
+
if (provider === "cloudflare-ai-gateway") {
|
|
225
|
+
const existingToken = parseCloudflareAiGatewayCredential(existing.key)?.token;
|
|
226
|
+
const incomingToken = parseCloudflareAiGatewayCredential(incoming.key)?.token;
|
|
227
|
+
return existingToken !== undefined && existingToken === incomingToken;
|
|
228
|
+
}
|
|
229
|
+
return false;
|
|
222
230
|
}
|
|
223
231
|
const incomingIdentifiers = extractOAuthCredentialIdentifiers(incoming);
|
|
224
232
|
const incomingIdentityKey = resolveProviderCredentialIdentityKey(provider, incomingIdentifiers);
|
|
@@ -208,7 +208,7 @@ const usageAmountSchema = type({
|
|
|
208
208
|
"remaining?": "number",
|
|
209
209
|
"usedFraction?": "number",
|
|
210
210
|
"remainingFraction?": "number",
|
|
211
|
-
unit: "'percent' | 'tokens' | 'requests' | 'usd' | 'minutes' | 'bytes' | 'unknown'",
|
|
211
|
+
unit: "'percent' | 'tokens' | 'requests' | 'credits' | 'usd' | 'minutes' | 'bytes' | 'unknown'",
|
|
212
212
|
});
|
|
213
213
|
|
|
214
214
|
const usageScopeSchema = type({
|
package/src/auth-storage.ts
CHANGED
|
@@ -28,6 +28,7 @@ import type {
|
|
|
28
28
|
OAuthProvider,
|
|
29
29
|
OAuthProviderId,
|
|
30
30
|
} from "./registry/oauth/types";
|
|
31
|
+
import { AUTHENTICATED_SENTINEL } from "./registry/types";
|
|
31
32
|
import { getEnvApiKey, getEnvApiKeyName } from "./stream";
|
|
32
33
|
import type { Provider } from "./types";
|
|
33
34
|
import type {
|
|
@@ -1272,6 +1273,7 @@ type UsageRankedCandidate<T extends AuthCredential> = UsageCandidate<T> & {
|
|
|
1272
1273
|
blocked: boolean;
|
|
1273
1274
|
blockedUntil?: number;
|
|
1274
1275
|
hasPriorityBoost: boolean;
|
|
1276
|
+
usageMeasured: boolean;
|
|
1275
1277
|
planPriority: number;
|
|
1276
1278
|
secondaryUsed: number;
|
|
1277
1279
|
secondaryRequiredDrain: number;
|
|
@@ -2168,13 +2170,16 @@ export class AuthStorage {
|
|
|
2168
2170
|
const windows = usage ? strategy.findWindowLimits(usage, args.rankingContext) : undefined;
|
|
2169
2171
|
const primary = windows?.primary;
|
|
2170
2172
|
const secondary = windows?.secondary;
|
|
2173
|
+
const usageMeasured = primary !== undefined || secondary !== undefined;
|
|
2174
|
+
const primaryUncapped = primary === undefined && secondary !== undefined;
|
|
2171
2175
|
ranked.push({
|
|
2172
2176
|
selection,
|
|
2173
2177
|
usage,
|
|
2174
2178
|
usageChecked,
|
|
2175
2179
|
blocked,
|
|
2176
2180
|
blockedUntil,
|
|
2177
|
-
|
|
2181
|
+
usageMeasured,
|
|
2182
|
+
hasPriorityBoost: strategy.hasPriorityBoost?.(primary, primaryUncapped, args.rankingContext) ?? false,
|
|
2178
2183
|
planPriority: 0,
|
|
2179
2184
|
secondaryUsed: this.#normalizeUsageFraction(secondary),
|
|
2180
2185
|
secondaryRequiredDrain: this.#computeWindowRequiredDrain(
|
|
@@ -2748,6 +2753,34 @@ export class AuthStorage {
|
|
|
2748
2753
|
return false;
|
|
2749
2754
|
}
|
|
2750
2755
|
|
|
2756
|
+
/**
|
|
2757
|
+
* Like {@link hasAuth} but excludes providers whose only credential is the
|
|
2758
|
+
* self-resolving {@link AUTHENTICATED_SENTINEL} — the marker AWS/Vertex
|
|
2759
|
+
* transports return when a credential *source* merely exists (a stray
|
|
2760
|
+
* `~/.aws` profile, an EC2 instance role, Application Default Credentials)
|
|
2761
|
+
* without a usable key resolved yet. Default-model auto-selection uses this
|
|
2762
|
+
* so an ambiently-available provider (e.g. `amazon-bedrock` via an unrelated
|
|
2763
|
+
* AWS profile) does not win the startup default over a provider the user
|
|
2764
|
+
* actually signed into and then 403 on the first turn. Explicit selection
|
|
2765
|
+
* and picker visibility still go through {@link hasAuth}. See issue #9967.
|
|
2766
|
+
*/
|
|
2767
|
+
hasConcreteAuth(provider: string): boolean {
|
|
2768
|
+
if (this.#runtimeOverrides.has(provider)) return true;
|
|
2769
|
+
if (this.#configOverrides.has(provider)) return true;
|
|
2770
|
+
if (this.#getCredentialsForProvider(provider).length > 0) return true;
|
|
2771
|
+
if ((provider === "amazon-bedrock" || provider === "bedrock-mantle") && $env.AWS_BEARER_TOKEN_BEDROCK?.trim()) {
|
|
2772
|
+
return true;
|
|
2773
|
+
}
|
|
2774
|
+
if (provider === "xai-oauth") {
|
|
2775
|
+
if ($env.XAI_OAUTH_TOKEN?.trim()) return true;
|
|
2776
|
+
} else {
|
|
2777
|
+
const envApiKey = getEnvApiKey(provider);
|
|
2778
|
+
if (envApiKey !== undefined && envApiKey !== AUTHENTICATED_SENTINEL) return true;
|
|
2779
|
+
}
|
|
2780
|
+
const fallback = this.#fallbackResolver?.(provider);
|
|
2781
|
+
return fallback !== undefined && fallback !== AUTHENTICATED_SENTINEL;
|
|
2782
|
+
}
|
|
2783
|
+
|
|
2751
2784
|
/**
|
|
2752
2785
|
* Whether a request could resolve a key for this provider, including
|
|
2753
2786
|
* cross-provider env aliases (`xai-oauth` borrowing `XAI_API_KEY`).
|
|
@@ -4633,8 +4666,8 @@ export class AuthStorage {
|
|
|
4633
4666
|
// scores are only comparable between measured windows, and the
|
|
4634
4667
|
// clockless headroom fallback (0..1) must not let an account whose
|
|
4635
4668
|
// usage fetch failed shadow a measured sibling.
|
|
4636
|
-
const leftMeasured = left.
|
|
4637
|
-
const rightMeasured = right.
|
|
4669
|
+
const leftMeasured = left.usageMeasured;
|
|
4670
|
+
const rightMeasured = right.usageMeasured;
|
|
4638
4671
|
if (leftMeasured !== rightMeasured) return leftMeasured ? -1 : 1;
|
|
4639
4672
|
// Required drain, descending: the account whose remaining quota must
|
|
4640
4673
|
// burn fastest to avoid expiring unused at its reset comes first, so
|
|
@@ -4769,13 +4802,16 @@ export class AuthStorage {
|
|
|
4769
4802
|
const windows = usage ? strategy.findWindowLimits(usage, args.rankingContext) : undefined;
|
|
4770
4803
|
const primary = windows?.primary;
|
|
4771
4804
|
const secondary = windows?.secondary;
|
|
4805
|
+
const usageMeasured = primary !== undefined || secondary !== undefined;
|
|
4806
|
+
const primaryUncapped = primary === undefined && secondary !== undefined;
|
|
4772
4807
|
ranked.push({
|
|
4773
4808
|
selection,
|
|
4774
4809
|
usage,
|
|
4775
4810
|
usageChecked,
|
|
4776
4811
|
blocked,
|
|
4777
4812
|
blockedUntil,
|
|
4778
|
-
|
|
4813
|
+
usageMeasured,
|
|
4814
|
+
hasPriorityBoost: strategy.hasPriorityBoost?.(primary, primaryUncapped, args.rankingContext) ?? false,
|
|
4779
4815
|
planPriority: getOpenAICodexPlanPriority(usage, args.planRequirement),
|
|
4780
4816
|
secondaryUsed: this.#normalizeUsageFraction(secondary),
|
|
4781
4817
|
secondaryRequiredDrain: this.#computeWindowRequiredDrain(
|
|
@@ -1089,6 +1089,15 @@ function buildAdditionalModelRequestFields(
|
|
|
1089
1089
|
};
|
|
1090
1090
|
}
|
|
1091
1091
|
|
|
1092
|
+
if (mode === "effort") {
|
|
1093
|
+
// OpenAI-schema models on Bedrock (the GPT-5.x SKUs) reject the
|
|
1094
|
+
// Anthropic budget block with `unknown_parameter: 'thinking'` and take
|
|
1095
|
+
// `reasoning.effort` instead — same effort vocabulary the catalog
|
|
1096
|
+
// already bakes (low/medium/high/xhigh/max).
|
|
1097
|
+
const level = requireSupportedEffort(model, reasoning);
|
|
1098
|
+
return { reasoning: { effort: model.thinking?.effortMap?.[level] ?? level } };
|
|
1099
|
+
}
|
|
1100
|
+
|
|
1092
1101
|
const level = requireSupportedEffort(model, reasoning);
|
|
1093
1102
|
const defaultBudgets: Record<Effort, number> = {
|
|
1094
1103
|
minimal: 1024,
|
package/src/providers/cursor.ts
CHANGED
|
@@ -4844,6 +4844,27 @@ export function buildCursorSystemPromptJsons(systemPrompt: readonly string[] | u
|
|
|
4844
4844
|
return systemPrompts.map(content => JSON.stringify({ role: "system", content }));
|
|
4845
4845
|
}
|
|
4846
4846
|
|
|
4847
|
+
function collectCursorToolHistory(messages: Message[], historyEnd: number) {
|
|
4848
|
+
const toolResults = new Map<string, ToolResultMessage>();
|
|
4849
|
+
const pairedToolCallIds = new Set<string>();
|
|
4850
|
+
for (let index = 0; index < historyEnd; index++) {
|
|
4851
|
+
const message = messages[index];
|
|
4852
|
+
if (message.role === "toolResult") {
|
|
4853
|
+
toolResults.set(message.toolCallId, message);
|
|
4854
|
+
} else if (message.role === "assistant") {
|
|
4855
|
+
for (const item of message.content) {
|
|
4856
|
+
if (item.type === "toolCall") pairedToolCallIds.add(item.id);
|
|
4857
|
+
}
|
|
4858
|
+
}
|
|
4859
|
+
}
|
|
4860
|
+
return { toolResults, pairedToolCallIds };
|
|
4861
|
+
}
|
|
4862
|
+
|
|
4863
|
+
function cursorOrphanToolResultText(result: ToolResultMessage): string {
|
|
4864
|
+
const prefix = result.isError ? "[Tool Error]" : "[Tool Result]";
|
|
4865
|
+
return `${prefix}\n${toolResultToText(result) || "(empty result)"}`;
|
|
4866
|
+
}
|
|
4867
|
+
|
|
4847
4868
|
function buildRootPromptMessagesJson(
|
|
4848
4869
|
messages: Message[],
|
|
4849
4870
|
systemPromptIds: Uint8Array[],
|
|
@@ -4852,6 +4873,8 @@ function buildRootPromptMessagesJson(
|
|
|
4852
4873
|
targetModelId?: string,
|
|
4853
4874
|
): Uint8Array[] {
|
|
4854
4875
|
assertCursorKimiK3HistoryReplayable(messages, activeUserMessageIndex, targetModelId);
|
|
4876
|
+
const historyEnd = activeUserMessageIndex >= 0 ? activeUserMessageIndex : messages.length;
|
|
4877
|
+
const { pairedToolCallIds } = collectCursorToolHistory(messages, historyEnd);
|
|
4855
4878
|
const entries: Uint8Array[] = [...systemPromptIds];
|
|
4856
4879
|
const pushJson = (obj: unknown) => {
|
|
4857
4880
|
const bytes = new TextEncoder().encode(JSON.stringify(obj));
|
|
@@ -4870,6 +4893,13 @@ function buildRootPromptMessagesJson(
|
|
|
4870
4893
|
if (content.length === 0) continue;
|
|
4871
4894
|
pushJson({ role: "assistant", content });
|
|
4872
4895
|
} else if (msg.role === "toolResult") {
|
|
4896
|
+
if (!pairedToolCallIds.has(msg.toolCallId)) {
|
|
4897
|
+
pushJson({
|
|
4898
|
+
role: "assistant",
|
|
4899
|
+
content: [{ type: "text", text: cursorOrphanToolResultText(msg) }],
|
|
4900
|
+
});
|
|
4901
|
+
continue;
|
|
4902
|
+
}
|
|
4873
4903
|
// Emit even when the result text is empty: the assistant `tool-call` is
|
|
4874
4904
|
// already in history, so dropping the pair would replay an orphaned call.
|
|
4875
4905
|
const toolCallId = normalizeToolCallId(msg.toolCallId);
|
|
@@ -5008,18 +5038,7 @@ function buildConversationTurns(
|
|
|
5008
5038
|
): Uint8Array[] {
|
|
5009
5039
|
const turns: Uint8Array[] = [];
|
|
5010
5040
|
const historyEnd = activeUserMessageIndex >= 0 ? activeUserMessageIndex : messages.length;
|
|
5011
|
-
const toolResults =
|
|
5012
|
-
const pairedToolCallIds = new Set<string>();
|
|
5013
|
-
for (let index = 0; index < historyEnd; index++) {
|
|
5014
|
-
const message = messages[index];
|
|
5015
|
-
if (message.role === "toolResult") {
|
|
5016
|
-
toolResults.set(message.toolCallId, message);
|
|
5017
|
-
} else if (message.role === "assistant") {
|
|
5018
|
-
for (const item of message.content) {
|
|
5019
|
-
if (item.type === "toolCall") pairedToolCallIds.add(item.id);
|
|
5020
|
-
}
|
|
5021
|
-
}
|
|
5022
|
-
}
|
|
5041
|
+
const { toolResults, pairedToolCallIds } = collectCursorToolHistory(messages, historyEnd);
|
|
5023
5042
|
|
|
5024
5043
|
let i = 0;
|
|
5025
5044
|
while (i < messages.length) {
|
|
@@ -5077,17 +5096,13 @@ function buildConversationTurns(
|
|
|
5077
5096
|
stepBlobIds.push(storeCursorBlob(blobStore, toBinary(ConversationStepSchema, step)));
|
|
5078
5097
|
}
|
|
5079
5098
|
} else if (stepMsg.role === "toolResult" && !pairedToolCallIds.has(stepMsg.toolCallId)) {
|
|
5080
|
-
const
|
|
5081
|
-
|
|
5082
|
-
|
|
5083
|
-
|
|
5084
|
-
|
|
5085
|
-
|
|
5086
|
-
|
|
5087
|
-
},
|
|
5088
|
-
});
|
|
5089
|
-
stepBlobIds.push(storeCursorBlob(blobStore, toBinary(ConversationStepSchema, step)));
|
|
5090
|
-
}
|
|
5099
|
+
const step = create(ConversationStepSchema, {
|
|
5100
|
+
message: {
|
|
5101
|
+
case: "assistantMessage",
|
|
5102
|
+
value: create(AssistantMessageSchema, { text: cursorOrphanToolResultText(stepMsg) }),
|
|
5103
|
+
},
|
|
5104
|
+
});
|
|
5105
|
+
stepBlobIds.push(storeCursorBlob(blobStore, toBinary(ConversationStepSchema, step)));
|
|
5091
5106
|
}
|
|
5092
5107
|
i++;
|
|
5093
5108
|
}
|
|
@@ -1,26 +1,130 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import
|
|
1
|
+
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
|
2
|
+
import {
|
|
3
|
+
CLOUDFLARE_AI_GATEWAY_ANTHROPIC_BASE_URL,
|
|
4
|
+
CLOUDFLARE_AI_GATEWAY_BASE_URL,
|
|
5
|
+
CLOUDFLARE_AI_GATEWAY_COMPAT_BASE_URL,
|
|
6
|
+
CLOUDFLARE_AI_GATEWAY_OPENAI_BASE_URL,
|
|
7
|
+
parseCloudflareAiGatewayCredential,
|
|
8
|
+
serializeCloudflareAiGatewayCredential,
|
|
9
|
+
} from "@oh-my-pi/pi-catalog/wire/cloudflare-ai-gateway";
|
|
10
|
+
import { $env } from "@oh-my-pi/pi-utils";
|
|
11
|
+
import * as AIError from "../error";
|
|
12
|
+
import { NO_AUTH_SENTINEL } from "../providers/openai-shared";
|
|
13
|
+
import type { OAuthController, OAuthLoginCallbacks } from "./oauth/types";
|
|
3
14
|
import type { ProviderDefinition } from "./types";
|
|
4
15
|
|
|
5
16
|
const AUTH_URL = "https://developers.cloudflare.com/ai-gateway/configuration/authentication/";
|
|
6
17
|
|
|
7
|
-
/**
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
});
|
|
18
|
+
/** Collect the gateway credential used by CLI, setup-wizard, and TUI login callers. */
|
|
19
|
+
export async function loginCloudflareAiGateway(options: OAuthController): Promise<string> {
|
|
20
|
+
if (!options.onPrompt) {
|
|
21
|
+
throw new AIError.OnPromptRequiredError("Cloudflare AI Gateway");
|
|
22
|
+
}
|
|
23
|
+
options.onAuth?.({
|
|
24
|
+
url: AUTH_URL,
|
|
25
|
+
instructions: "Create an AI Gateway token with Run permission, then copy it here.",
|
|
26
|
+
});
|
|
27
|
+
|
|
28
|
+
const apiKey = await options.onPrompt({
|
|
29
|
+
message: "Paste your Cloudflare AI Gateway token/API key",
|
|
30
|
+
placeholder: "cfut_...",
|
|
31
|
+
});
|
|
32
|
+
if (options.signal?.aborted) throw new AIError.LoginCancelledError();
|
|
33
|
+
if (!apiKey.trim()) throw new AIError.ApiKeyRequiredError();
|
|
34
|
+
|
|
35
|
+
const accountId = await options.onPrompt({
|
|
36
|
+
message: "Enter your Cloudflare account ID",
|
|
37
|
+
placeholder: "32-character account ID",
|
|
38
|
+
});
|
|
39
|
+
if (options.signal?.aborted) throw new AIError.LoginCancelledError();
|
|
40
|
+
if (!accountId.trim()) throw new AIError.ConfigurationError("Cloudflare account ID is required");
|
|
41
|
+
|
|
42
|
+
const gatewayId = await options.onPrompt({
|
|
43
|
+
message: "Enter your Cloudflare AI Gateway ID",
|
|
44
|
+
placeholder: "default",
|
|
45
|
+
});
|
|
46
|
+
if (options.signal?.aborted) throw new AIError.LoginCancelledError();
|
|
47
|
+
if (!gatewayId.trim()) throw new AIError.ConfigurationError("Cloudflare AI Gateway ID is required");
|
|
48
|
+
|
|
49
|
+
return serializeCloudflareAiGatewayCredential(apiKey, accountId, gatewayId);
|
|
50
|
+
}
|
|
21
51
|
|
|
22
52
|
export const cloudflareAiGatewayProvider = {
|
|
23
53
|
id: "cloudflare-ai-gateway",
|
|
24
54
|
name: "Cloudflare AI Gateway",
|
|
55
|
+
prepareModel: model => {
|
|
56
|
+
const hasGatewayPlaceholders = model.baseUrl.includes("<account>") || model.baseUrl.includes("<gateway>");
|
|
57
|
+
if (model.id.startsWith("anthropic/")) {
|
|
58
|
+
const requestModelId = model.id.slice("anthropic/".length).replaceAll(".", "-");
|
|
59
|
+
const baseUrl = hasGatewayPlaceholders ? CLOUDFLARE_AI_GATEWAY_ANTHROPIC_BASE_URL : model.baseUrl;
|
|
60
|
+
if (
|
|
61
|
+
model.api === "anthropic-messages" &&
|
|
62
|
+
model.baseUrl === baseUrl &&
|
|
63
|
+
model.requestModelId === requestModelId
|
|
64
|
+
) {
|
|
65
|
+
return model;
|
|
66
|
+
}
|
|
67
|
+
return {
|
|
68
|
+
...model,
|
|
69
|
+
api: "anthropic-messages",
|
|
70
|
+
baseUrl,
|
|
71
|
+
requestModelId,
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
if (model.id.startsWith("openai/")) {
|
|
75
|
+
const requestModelId = model.id.slice("openai/".length);
|
|
76
|
+
const baseUrl = hasGatewayPlaceholders ? CLOUDFLARE_AI_GATEWAY_OPENAI_BASE_URL : model.baseUrl;
|
|
77
|
+
if (model.api === "openai-responses" && model.baseUrl === baseUrl && model.requestModelId === requestModelId) {
|
|
78
|
+
return model;
|
|
79
|
+
}
|
|
80
|
+
return buildModel({
|
|
81
|
+
...model,
|
|
82
|
+
api: "openai-responses",
|
|
83
|
+
baseUrl,
|
|
84
|
+
compat: model.compatConfig,
|
|
85
|
+
requestModelId,
|
|
86
|
+
});
|
|
87
|
+
}
|
|
88
|
+
if (model.id.startsWith("workers-ai/")) {
|
|
89
|
+
const baseUrl = hasGatewayPlaceholders ? CLOUDFLARE_AI_GATEWAY_COMPAT_BASE_URL : model.baseUrl;
|
|
90
|
+
if (model.api === "openai-completions" && model.baseUrl === baseUrl) {
|
|
91
|
+
return model;
|
|
92
|
+
}
|
|
93
|
+
return buildModel({
|
|
94
|
+
...model,
|
|
95
|
+
api: "openai-completions",
|
|
96
|
+
baseUrl,
|
|
97
|
+
compat: model.compatConfig,
|
|
98
|
+
});
|
|
99
|
+
}
|
|
100
|
+
return model;
|
|
101
|
+
},
|
|
102
|
+
prepareRequest: (model, options) => {
|
|
103
|
+
const credential = parseCloudflareAiGatewayCredential(options.apiKey ?? $env.CLOUDFLARE_AI_GATEWAY_API_KEY ?? "");
|
|
104
|
+
if (!credential) return { model, options };
|
|
105
|
+
const accountId = credential.accountId ?? $env.CLOUDFLARE_ACCOUNT_ID;
|
|
106
|
+
const gatewayId = credential.gatewayId ?? $env.CLOUDFLARE_GATEWAY_ID;
|
|
107
|
+
let baseUrl = model.baseUrl;
|
|
108
|
+
if (baseUrl.startsWith(CLOUDFLARE_AI_GATEWAY_BASE_URL)) {
|
|
109
|
+
if (!accountId) throw new AIError.ConfigurationError("Cloudflare account ID is required");
|
|
110
|
+
if (!gatewayId) throw new AIError.ConfigurationError("Cloudflare AI Gateway ID is required");
|
|
111
|
+
baseUrl = baseUrl.replace("<account>", accountId).replace("<gateway>", gatewayId);
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const isAnthropic = model.api === "anthropic-messages";
|
|
115
|
+
let headers = model.headers;
|
|
116
|
+
if (!isAnthropic) {
|
|
117
|
+
headers = { ...headers };
|
|
118
|
+
for (const name in headers) {
|
|
119
|
+
const normalized = name.toLowerCase();
|
|
120
|
+
if (normalized === "authorization" || normalized === "x-api-key") delete headers[name];
|
|
121
|
+
}
|
|
122
|
+
headers["cf-aig-authorization"] = `Bearer ${credential.token}`;
|
|
123
|
+
}
|
|
124
|
+
return {
|
|
125
|
+
model: { ...model, baseUrl, headers },
|
|
126
|
+
options: { ...options, apiKey: isAnthropic ? credential.token : NO_AUTH_SENTINEL },
|
|
127
|
+
};
|
|
128
|
+
},
|
|
25
129
|
login: (cb: OAuthLoginCallbacks) => loginCloudflareAiGateway(cb),
|
|
26
130
|
} as const satisfies ProviderDefinition;
|
|
@@ -303,10 +303,16 @@
|
|
|
303
303
|
const message = document.getElementById("message");
|
|
304
304
|
|
|
305
305
|
if (serverState.ok) {
|
|
306
|
+
const closeButton = document.querySelector(".btn");
|
|
306
307
|
app.classList.add("success", "countdown");
|
|
307
308
|
title.textContent = "Authentication Successful";
|
|
308
|
-
message.
|
|
309
|
-
|
|
309
|
+
message.textContent = "You have successfully logged in.";
|
|
310
|
+
window.close();
|
|
311
|
+
setTimeout(() => {
|
|
312
|
+
app.classList.remove("countdown");
|
|
313
|
+
closeButton.remove();
|
|
314
|
+
message.innerHTML = "You have successfully logged in.<br>Please close this tab manually.";
|
|
315
|
+
}, 300);
|
|
310
316
|
} else {
|
|
311
317
|
app.classList.add("error");
|
|
312
318
|
title.textContent = "Authentication Failed";
|
package/src/registry/types.ts
CHANGED
|
@@ -29,6 +29,7 @@ export interface PreparedProviderRequest {
|
|
|
29
29
|
}
|
|
30
30
|
|
|
31
31
|
export type ProviderRequestPreparer = (model: Model<Api>, options: StreamOptions) => PreparedProviderRequest;
|
|
32
|
+
export type ProviderModelPreparer = (model: Model<Api>) => Model<Api>;
|
|
32
33
|
export type ProviderSimpleOptionsMapper = (options: SimpleStreamOptions) => Readonly<Record<string, unknown>>;
|
|
33
34
|
|
|
34
35
|
export interface ProviderModelDiscoveryConfig {
|
|
@@ -66,6 +67,8 @@ export interface ProviderDefinition {
|
|
|
66
67
|
readonly envKeys?: KeyResolver;
|
|
67
68
|
/** Provider transport can authenticate without a resolved API-key string. */
|
|
68
69
|
readonly allowsMissingApiKey?: boolean;
|
|
70
|
+
/** Provider-owned model normalization that must run before API-specific option mapping. */
|
|
71
|
+
readonly prepareModel?: ProviderModelPreparer;
|
|
69
72
|
/** Provider-owned request shaping applied before generic API dispatch. */
|
|
70
73
|
readonly prepareRequest?: ProviderRequestPreparer;
|
|
71
74
|
/** Provider-owned projection from the generic simple-stream option bag. */
|
package/src/stream.ts
CHANGED
|
@@ -939,9 +939,10 @@ function streamDispatch<TApi extends Api>(
|
|
|
939
939
|
return streamBedrock(model as Model<"bedrock-converse-stream">, context, requestOptions as BedrockOptions);
|
|
940
940
|
}
|
|
941
941
|
|
|
942
|
-
const
|
|
943
|
-
const
|
|
944
|
-
const
|
|
942
|
+
const providerDefinition = getProviderDefinition(model.provider);
|
|
943
|
+
const requestModel = providerDefinition?.prepareModel?.(model) ?? model;
|
|
944
|
+
const prepared = providerDefinition?.prepareRequest?.(requestModel, requestOptions as StreamOptions);
|
|
945
|
+
const providerModel = prepared?.model ?? requestModel;
|
|
945
946
|
const preparedOptions = prepared?.options ?? (requestOptions as StreamOptions);
|
|
946
947
|
const apiKey = preparedOptions.apiKey || getEnvApiKey(providerModel.provider);
|
|
947
948
|
if (!apiKey) {
|
|
@@ -1691,8 +1692,9 @@ function streamSimpleRequest<TApi extends Api>(
|
|
|
1691
1692
|
),
|
|
1692
1693
|
);
|
|
1693
1694
|
}
|
|
1694
|
-
const
|
|
1695
|
-
|
|
1695
|
+
const providerModel = getProviderDefinition(model.provider)?.prepareModel?.(model) ?? model;
|
|
1696
|
+
const providerOptions = mapOptionsForApi(providerModel, requestOptions, apiKey);
|
|
1697
|
+
return stream(providerModel, context, providerOptions);
|
|
1696
1698
|
}
|
|
1697
1699
|
|
|
1698
1700
|
export async function completeSimple<TApi extends Api>(
|
|
@@ -2080,8 +2082,8 @@ function mapOptionsForApi<TApi extends Api>(
|
|
|
2080
2082
|
guardrailVersion: model.guardrailVersion ?? options?.guardrailVersion,
|
|
2081
2083
|
guardrailTrace: model.guardrailTrace ?? options?.guardrailTrace,
|
|
2082
2084
|
};
|
|
2083
|
-
//
|
|
2084
|
-
if (model.thinking?.mode === "anthropic-adaptive") {
|
|
2085
|
+
// Effort modes send effort directly, no budget_tokens — skip budget inflation.
|
|
2086
|
+
if (model.thinking?.mode === "effort" || model.thinking?.mode === "anthropic-adaptive") {
|
|
2085
2087
|
return castApi<"bedrock-converse-stream">(bedrockBase);
|
|
2086
2088
|
}
|
|
2087
2089
|
const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
|
package/src/types.ts
CHANGED
|
@@ -879,6 +879,10 @@ export interface DeveloperMessage {
|
|
|
879
879
|
content: string | (TextContent | ImageContent)[];
|
|
880
880
|
/** Who initiated this message for billing/attribution semantics. */
|
|
881
881
|
attribution?: MessageAttribution;
|
|
882
|
+
/** True if the message was injected by the system (e.g., auto-continue) and initiates a fresh run rather than continuing the current one. */
|
|
883
|
+
synthetic?: boolean;
|
|
884
|
+
/** True when the synthetic prompt was a deliberate operator action (`.`, `c` continue shortcut) rather than an automatic continuation — its timestamp is the turn's prompt time. */
|
|
885
|
+
userInitiated?: boolean;
|
|
882
886
|
/** Provider-specific opaque payload used to reconstruct transport-native history. */
|
|
883
887
|
providerPayload?: ProviderPayload;
|
|
884
888
|
timestamp: number; // Unix timestamp in milliseconds
|
|
@@ -976,6 +980,8 @@ export interface AssistantMessage {
|
|
|
976
980
|
timestamp: number; // Unix timestamp in milliseconds
|
|
977
981
|
duration?: number; // Request duration in milliseconds
|
|
978
982
|
ttft?: number; // Time to first token in milliseconds
|
|
983
|
+
/** Local wall-clock time the response finished streaming (ms since epoch); stamped by the session at message_end so prompt→yield timing never depends on provider-reported duration. */
|
|
984
|
+
completedAt?: number;
|
|
979
985
|
}
|
|
980
986
|
|
|
981
987
|
export interface ToolResultMessage<TDetails = unknown> {
|
|
@@ -426,12 +426,19 @@ export const antigravityUsageProvider: UsageProvider = {
|
|
|
426
426
|
supports: params => params.provider === "google-antigravity",
|
|
427
427
|
};
|
|
428
428
|
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
if (
|
|
433
|
-
if (
|
|
434
|
-
if (
|
|
429
|
+
/** Map an Antigravity model id to its backend quota-counter key. */
|
|
430
|
+
export function getAntigravityCounterKeyForModel(modelId: string | undefined): string | undefined {
|
|
431
|
+
const normalizedModelId = modelId?.toLowerCase();
|
|
432
|
+
if (!normalizedModelId) return undefined;
|
|
433
|
+
if (normalizedModelId.startsWith("claude-")) return "anthropic";
|
|
434
|
+
if (
|
|
435
|
+
normalizedModelId.startsWith("gemini-") ||
|
|
436
|
+
normalizedModelId.startsWith("gemma-") ||
|
|
437
|
+
normalizedModelId.startsWith("tab_")
|
|
438
|
+
) {
|
|
439
|
+
return "google";
|
|
440
|
+
}
|
|
441
|
+
if (normalizedModelId.startsWith("gpt-") || normalizedModelId.startsWith("openai/")) return "openai";
|
|
435
442
|
return undefined;
|
|
436
443
|
}
|
|
437
444
|
|
|
@@ -440,14 +447,19 @@ function getAntigravityCounterLimits(report: UsageReport, counterKey: string): U
|
|
|
440
447
|
return report.limits.filter(limit => limit.id.toLowerCase().startsWith(prefix));
|
|
441
448
|
}
|
|
442
449
|
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
450
|
+
/**
|
|
451
|
+
* Scope an Antigravity report to the active model's backend counter, falling
|
|
452
|
+
* back to legacy default counters only when that backend has no limits.
|
|
453
|
+
*
|
|
454
|
+
* Exhaustion checks are only safe with a concrete backend counter. A no-model
|
|
455
|
+
* credential lookup (for example image-provider discovery) must not turn one
|
|
456
|
+
* exhausted family into a provider-wide block.
|
|
457
|
+
*/
|
|
458
|
+
export function scopeAntigravityLimitsForModel(
|
|
447
459
|
report: UsageReport,
|
|
448
460
|
context: CredentialRankingContext | undefined,
|
|
449
461
|
): UsageLimit[] {
|
|
450
|
-
const counterKey = getAntigravityCounterKeyForModel(context);
|
|
462
|
+
const counterKey = getAntigravityCounterKeyForModel(context?.modelId);
|
|
451
463
|
if (!counterKey) return [];
|
|
452
464
|
const backendLimits = getAntigravityCounterLimits(report, counterKey);
|
|
453
465
|
if (backendLimits.length > 0) return backendLimits;
|
|
@@ -455,7 +467,7 @@ function scopeAntigravityLimitsForModel(
|
|
|
455
467
|
}
|
|
456
468
|
|
|
457
469
|
function rankAntigravityLimits(report: UsageReport, context: CredentialRankingContext | undefined): UsageLimit[] {
|
|
458
|
-
const counterKey = getAntigravityCounterKeyForModel(context);
|
|
470
|
+
const counterKey = getAntigravityCounterKeyForModel(context?.modelId);
|
|
459
471
|
if (!counterKey) return report.limits;
|
|
460
472
|
return scopeAntigravityLimitsForModel(report, context);
|
|
461
473
|
}
|
|
@@ -480,7 +492,7 @@ export const antigravityRankingStrategy: CredentialRankingStrategy = {
|
|
|
480
492
|
// Always return a scope for Antigravity so missing/unknown model context
|
|
481
493
|
// cannot fall through to AuthStorage's provider-wide block bucket.
|
|
482
494
|
blockScope(context) {
|
|
483
|
-
const counterKey = getAntigravityCounterKeyForModel(context);
|
|
495
|
+
const counterKey = getAntigravityCounterKeyForModel(context?.modelId);
|
|
484
496
|
return `counter:${counterKey ?? "unknown"}`;
|
|
485
497
|
},
|
|
486
498
|
// Antigravity windows carry `durationMs` when the response identifies them
|
|
@@ -611,8 +611,11 @@ export const codexRankingStrategy: CredentialRankingStrategy = {
|
|
|
611
611
|
return { primary: findLimit("primary"), secondary: findLimit("secondary") };
|
|
612
612
|
},
|
|
613
613
|
windowDefaults: { primaryMs: 60 * 60 * 1000, secondaryMs: 7 * 24 * 60 * 60 * 1000 },
|
|
614
|
-
hasPriorityBoost(primary) {
|
|
615
|
-
|
|
614
|
+
hasPriorityBoost(primary, primaryUncapped = false, context) {
|
|
615
|
+
// Chat plans can omit an uncapped primary window while retaining their
|
|
616
|
+
// weekly window. Spark always has a capped primary meter, so a missing
|
|
617
|
+
// Spark primary is incomplete rather than uncapped.
|
|
618
|
+
if (!primary) return primaryUncapped && !isCodexSparkRequest(context);
|
|
616
619
|
const windowId = primary.scope.windowId?.toLowerCase();
|
|
617
620
|
const durationMs = primary.window?.durationMs;
|
|
618
621
|
const isFiveHourWindow =
|
package/src/usage/zai.ts
CHANGED
|
@@ -51,6 +51,8 @@ interface ZaiQuotaPayload {
|
|
|
51
51
|
msg?: string;
|
|
52
52
|
data?: {
|
|
53
53
|
limits?: ZaiUsageLimitItem[];
|
|
54
|
+
/** Coding-plan tier (e.g. "lite", "pro", "max") surfaced as the plan label. */
|
|
55
|
+
level?: string;
|
|
54
56
|
};
|
|
55
57
|
}
|
|
56
58
|
|
|
@@ -193,17 +195,33 @@ function buildModelUsageUrl(baseUrl: string, now: Date): string {
|
|
|
193
195
|
}
|
|
194
196
|
|
|
195
197
|
function getZaiCredentialLimits(report: UsageReport): UsageLimit[] {
|
|
196
|
-
|
|
197
|
-
limit =>
|
|
198
|
+
return report.limits.filter(
|
|
199
|
+
limit =>
|
|
200
|
+
limit.id.startsWith("zai:requests:") ||
|
|
201
|
+
limit.id.startsWith("zai:tokens:") ||
|
|
202
|
+
limit.id.startsWith("zai:credits:"),
|
|
198
203
|
);
|
|
199
|
-
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
function zaiLimitPressure(limit: UsageLimit): number {
|
|
207
|
+
const fraction = limit.amount.usedFraction;
|
|
208
|
+
return typeof fraction === "number" && Number.isFinite(fraction) ? fraction : -1;
|
|
200
209
|
}
|
|
201
210
|
|
|
202
211
|
function rankZaiRequestLimits(report: UsageReport): UsageLimit[] {
|
|
203
212
|
const requestLimits = report.limits.filter(limit => limit.id.startsWith("zai:requests:"));
|
|
204
213
|
const credentialLimits = getZaiCredentialLimits(report);
|
|
205
214
|
const limits = requestLimits.length > 0 ? requestLimits : credentialLimits;
|
|
206
|
-
|
|
215
|
+
// Mixed-meter payloads (tokens + credits on the same plan) can repeat a
|
|
216
|
+
// window; keep the most-binding limit per window so a second 5h row never
|
|
217
|
+
// displaces the weekly window when primary/secondary are picked positionally.
|
|
218
|
+
const byWindow = new Map<number, UsageLimit>();
|
|
219
|
+
for (const limit of limits) {
|
|
220
|
+
const durationMs = limit.window?.durationMs ?? Number.POSITIVE_INFINITY;
|
|
221
|
+
const current = byWindow.get(durationMs);
|
|
222
|
+
if (!current || zaiLimitPressure(limit) > zaiLimitPressure(current)) byWindow.set(durationMs, limit);
|
|
223
|
+
}
|
|
224
|
+
const ranked = [...byWindow.values()];
|
|
207
225
|
ranked.sort((left, right) => {
|
|
208
226
|
const leftDuration = left.window?.durationMs ?? Number.POSITIVE_INFINITY;
|
|
209
227
|
const rightDuration = right.window?.durationMs ?? Number.POSITIVE_INFINITY;
|
|
@@ -306,6 +324,33 @@ async function fetchZaiUsage(params: UsageFetchParams, ctx: UsageFetchContext):
|
|
|
306
324
|
status: getUsageStatus(amount.usedFraction),
|
|
307
325
|
});
|
|
308
326
|
}
|
|
327
|
+
if (parsed.type === "CREDIT_LIMIT") {
|
|
328
|
+
// GLM Coding Plan windows (e.g. 12k credits / 5h + 60k credits / week):
|
|
329
|
+
// `usage` is the plan's credit allotment, `currentValue` the spend.
|
|
330
|
+
// `percentage` is a server-rounded integer (11 for 1438/12000 ≈ 11.98%),
|
|
331
|
+
// so prefer the exact ratio and fall back to it only without absolutes.
|
|
332
|
+
const window = buildZaiWindow(parsed);
|
|
333
|
+
const hasAbsoluteMeter = parsed.currentValue !== undefined && parsed.usage !== undefined && parsed.usage > 0;
|
|
334
|
+
const amount = buildUsageAmount({
|
|
335
|
+
used: parsed.currentValue,
|
|
336
|
+
limit: parsed.usage,
|
|
337
|
+
remaining: parsed.remaining,
|
|
338
|
+
percentage: hasAbsoluteMeter ? undefined : parsed.percentage,
|
|
339
|
+
unit: "credits",
|
|
340
|
+
});
|
|
341
|
+
limits.push({
|
|
342
|
+
id: `zai:credits:${window.id}`,
|
|
343
|
+
label: `ZAI ${window.label} Credit Quota`,
|
|
344
|
+
scope: {
|
|
345
|
+
provider: params.provider,
|
|
346
|
+
windowId: window.id,
|
|
347
|
+
shared: true,
|
|
348
|
+
},
|
|
349
|
+
window,
|
|
350
|
+
amount,
|
|
351
|
+
status: getUsageStatus(amount.usedFraction),
|
|
352
|
+
});
|
|
353
|
+
}
|
|
309
354
|
}
|
|
310
355
|
|
|
311
356
|
if (limits.length === 0) return null;
|
|
@@ -318,6 +363,7 @@ async function fetchZaiUsage(params: UsageFetchParams, ctx: UsageFetchContext):
|
|
|
318
363
|
endpoint: url,
|
|
319
364
|
accountId: credential.accountId,
|
|
320
365
|
email: credential.email,
|
|
366
|
+
...(typeof payload.data?.level === "string" && payload.data.level ? { planType: payload.data.level } : {}),
|
|
321
367
|
},
|
|
322
368
|
raw: payload,
|
|
323
369
|
};
|
package/src/usage.ts
CHANGED
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
*/
|
|
7
7
|
import { type } from "@oh-my-pi/omptype";
|
|
8
8
|
import type { FetchImpl, Provider } from "./types";
|
|
9
|
-
export type UsageUnit = "percent" | "tokens" | "requests" | "usd" | "minutes" | "bytes" | "unknown";
|
|
9
|
+
export type UsageUnit = "percent" | "tokens" | "requests" | "credits" | "usd" | "minutes" | "bytes" | "unknown";
|
|
10
10
|
|
|
11
11
|
export type UsageStatus = "ok" | "warning" | "exhausted" | "unknown";
|
|
12
12
|
|
|
@@ -240,7 +240,9 @@ export interface ClientUsageSummary {
|
|
|
240
240
|
|
|
241
241
|
// ─── Zod schemas (wire-shape validation for the broker `/v1/usage` endpoint) ─
|
|
242
242
|
|
|
243
|
-
export const usageUnitSchema = type(
|
|
243
|
+
export const usageUnitSchema = type(
|
|
244
|
+
"'percent' | 'tokens' | 'requests' | 'credits' | 'usd' | 'minutes' | 'bytes' | 'unknown'",
|
|
245
|
+
);
|
|
244
246
|
export const usageStatusSchema = type("'ok' | 'warning' | 'exhausted' | 'unknown'");
|
|
245
247
|
|
|
246
248
|
export const usageWindowSchema = type({
|
|
@@ -412,6 +414,14 @@ export interface CredentialRankingStrategy {
|
|
|
412
414
|
primaryMs: number;
|
|
413
415
|
secondaryMs: number;
|
|
414
416
|
};
|
|
415
|
-
/**
|
|
416
|
-
|
|
417
|
+
/**
|
|
418
|
+
* Optional: priority boost for specific credential states (e.g., fresh 5h
|
|
419
|
+
* ticker start). `primaryUncapped` is true only when the fetched report has
|
|
420
|
+
* an applicable secondary window but no applicable primary window.
|
|
421
|
+
*/
|
|
422
|
+
hasPriorityBoost?(
|
|
423
|
+
primary: UsageLimit | undefined,
|
|
424
|
+
primaryUncapped?: boolean,
|
|
425
|
+
context?: CredentialRankingContext,
|
|
426
|
+
): boolean;
|
|
417
427
|
}
|