@oh-my-pi/pi-ai 18.0.7 → 18.0.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,25 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.0.9] - 2026-08-28
6
+
7
+ ### Fixed
8
+
9
+ - Improved OAuth sign-in flows, including a fallback message when the browser cannot automatically close the OAuth success tab.
10
+ - Fixed Cloudflare AI Gateway onboarding and routing so gateway account and endpoint configuration is preserved correctly while gateway credentials are not sent as upstream OpenAI authorization headers.
11
+ - Fixed Codex OAuth quota handling so chat and Spark usage remain independent, legacy shared quota limits continue to work, and incomplete usage reports are not incorrectly treated as unlimited.
12
+
13
+ ## [18.0.8] - 2026-08-27
14
+
15
+ ### Added
16
+
17
+ - Added Z.AI GLM Coding Plan usage tracking: credit-based `CREDIT_LIMIT` windows (5h + weekly) now surface in `omp usage` and the status line with the plan tier (`plan: lite/pro/max`).
18
+
19
+ ### Fixed
20
+
21
+ - Fixed Amazon Bedrock requests to OpenAI-schema models (the `gpt-5.x` SKUs) failing with HTTP 400 `unknown_parameter: 'thinking'` when reasoning was enabled, by sending `reasoning.effort` instead of Anthropic's `thinking` budget block for models the catalog marks as effort-controlled.
22
+ - Fixed Cursor replay rejecting sessions with orphaned tool results while preserving their output as assistant context.
23
+
5
24
  ## [18.0.7] - 2026-08-26
6
25
 
7
26
  ### Added
package/README.md CHANGED
@@ -76,7 +76,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
76
76
  - **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
77
77
  - **QwenCloud Token Plan** (supports `/login alibaba-token-plan`, `ALIBABA_TOKEN_PLAN_API_KEY`, or `BAILIAN_TOKEN_PLAN_API_KEY`; interactive login first selects a region — International (Singapore, default), China (Beijing) for 百炼 Token Plan keys, or a custom base URL — since region keys are non-interchangeable, then optionally stores a `home.qwencloud.com` Cookie request header for best-effort 5-hour and 7-day quota reporting)
78
78
  To enable quota reporting, sign in to the Token Plan dashboard, copy the `Cookie` request-header value from a `home.qwencloud.com` request in browser developer tools, and paste it at the second login prompt. Press Enter to skip; the Cookie is sensitive and session-lived, so rerun login when it expires.
79
- - **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
79
+ - **Cloudflare AI Gateway** (supports `/login cloudflare-ai-gateway`, or `CLOUDFLARE_AI_GATEWAY_API_KEY` with `CLOUDFLARE_ACCOUNT_ID` and `CLOUDFLARE_GATEWAY_ID`)
80
80
  - **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
81
81
  - **Ollama Cloud** (hosted native Ollama API; requires `OLLAMA_CLOUD_API_KEY`)
82
82
  - **llama.cpp** (local OpenAI and Anthropic compatible inference server)
@@ -960,11 +960,10 @@ In Node.js environments, you can set environment variables to avoid passing API
960
960
  | Xiaomi MiMo | `XIAOMI_API_KEY` |
961
961
  | ZenMux | `ZENMUX_API_KEY` |
962
962
  | vLLM | `VLLM_API_KEY` |
963
- | Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` |
963
+ | Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` + `CLOUDFLARE_ACCOUNT_ID` + `CLOUDFLARE_GATEWAY_ID` |
964
964
  | GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
965
965
 
966
- For Cloudflare AI Gateway models, use provider base URL format
967
- `https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic`.
966
+ `/login cloudflare-ai-gateway` collects and stores the gateway token, account ID, and gateway ID. For environment configuration, set all three Cloudflare values above. OMP derives provider endpoints from the account and gateway IDs.
968
967
 
969
968
  For Anthropic Foundry routing, set `CLAUDE_CODE_USE_FOUNDRY=true` plus:
970
969
  `FOUNDRY_BASE_URL`, `ANTHROPIC_FOUNDRY_API_KEY`, optional `ANTHROPIC_CUSTOM_HEADERS`,
@@ -996,7 +995,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
996
995
  - Ollama: local OpenAI-compatible runtime (`http://127.0.0.1:11434/v1`)
997
996
  - Ollama Cloud: native Ollama API host (`https://ollama.com/api`, configured here as base URL `https://ollama.com`)
998
997
  - LiteLLM: `http://localhost:4000/v1`
999
- - Cloudflare AI Gateway: `https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic`
998
+ - Cloudflare AI Gateway: native Anthropic, OpenAI, and Workers AI routes under `https://gateway.ai.cloudflare.com/v1/<account>/<gateway>`
1000
999
  - Qwen Portal: `https://portal.qwen.ai/v1`
1001
1000
  When set, the library automatically uses these keys:
1002
1001
 
@@ -821,6 +821,18 @@ export declare class AuthStorage {
821
821
  * `XAI_API_KEY` does not auto-select SuperGrok (`xai-oauth`).
822
822
  */
823
823
  hasAuth(provider: string): boolean;
824
+ /**
825
+ * Like {@link hasAuth} but excludes providers whose only credential is the
826
+ * self-resolving {@link AUTHENTICATED_SENTINEL} — the marker AWS/Vertex
827
+ * transports return when a credential *source* merely exists (a stray
828
+ * `~/.aws` profile, an EC2 instance role, Application Default Credentials)
829
+ * without a usable key resolved yet. Default-model auto-selection uses this
830
+ * so an ambiently-available provider (e.g. `amazon-bedrock` via an unrelated
831
+ * AWS profile) does not win the startup default over a provider the user
832
+ * actually signed into and then 403 on the first turn. Explicit selection
833
+ * and picker visibility still go through {@link hasAuth}. See issue #9967.
834
+ */
835
+ hasConcreteAuth(provider: string): boolean;
824
836
  /**
825
837
  * Whether a request could resolve a key for this provider, including
826
838
  * cross-provider env aliases (`xai-oauth` borrowing `XAI_API_KEY`).
@@ -1,13 +1,13 @@
1
- import type { OAuthLoginCallbacks } from "./oauth/types.js";
2
- /**
3
- * Login to Cloudflare AI Gateway.
4
- *
5
- * Opens browser to Cloudflare AI Gateway authentication docs and prompts for a gateway token/API key.
6
- * Returns the API key directly (not OAuthCredentials - this isn't OAuth).
7
- */
8
- export declare const loginCloudflareAiGateway: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
1
+ import type { OAuthController, OAuthLoginCallbacks } from "./oauth/types.js";
2
+ /** Collect the gateway credential used by CLI, setup-wizard, and TUI login callers. */
3
+ export declare function loginCloudflareAiGateway(options: OAuthController): Promise<string>;
9
4
  export declare const cloudflareAiGatewayProvider: {
10
5
  readonly id: "cloudflare-ai-gateway";
11
6
  readonly name: "Cloudflare AI Gateway";
7
+ readonly prepareModel: (model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>) => import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>;
8
+ readonly prepareRequest: (model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>, options: import("../index.js").StreamOptions) => {
9
+ model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>;
10
+ options: import("../index.js").StreamOptions;
11
+ };
12
12
  readonly login: (cb: OAuthLoginCallbacks) => Promise<string>;
13
13
  };
@@ -72,6 +72,11 @@ declare const ALL: ({
72
72
  } | {
73
73
  readonly id: "cloudflare-ai-gateway";
74
74
  readonly name: "Cloudflare AI Gateway";
75
+ readonly prepareModel: (model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>) => import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>;
76
+ readonly prepareRequest: (model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>, options: import("../index.js").StreamOptions) => {
77
+ model: import("@oh-my-pi/pi-catalog").Model<import("@oh-my-pi/pi-catalog").Api>;
78
+ options: import("../index.js").StreamOptions;
79
+ };
75
80
  readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string>;
76
81
  } | {
77
82
  readonly id: "coreweave";
@@ -24,6 +24,7 @@ export interface PreparedProviderRequest {
24
24
  readonly options: StreamOptions;
25
25
  }
26
26
  export type ProviderRequestPreparer = (model: Model<Api>, options: StreamOptions) => PreparedProviderRequest;
27
+ export type ProviderModelPreparer = (model: Model<Api>) => Model<Api>;
27
28
  export type ProviderSimpleOptionsMapper = (options: SimpleStreamOptions) => Readonly<Record<string, unknown>>;
28
29
  export interface ProviderModelDiscoveryConfig {
29
30
  readonly apiKey?: string;
@@ -57,6 +58,8 @@ export interface ProviderDefinition {
57
58
  readonly envKeys?: KeyResolver;
58
59
  /** Provider transport can authenticate without a resolved API-key string. */
59
60
  readonly allowsMissingApiKey?: boolean;
61
+ /** Provider-owned model normalization that must run before API-specific option mapping. */
62
+ readonly prepareModel?: ProviderModelPreparer;
60
63
  /** Provider-owned request shaping applied before generic API dispatch. */
61
64
  readonly prepareRequest?: ProviderRequestPreparer;
62
65
  /** Provider-owned projection from the generic simple-stream option bag. */
@@ -698,6 +698,10 @@ export interface DeveloperMessage {
698
698
  content: string | (TextContent | ImageContent)[];
699
699
  /** Who initiated this message for billing/attribution semantics. */
700
700
  attribution?: MessageAttribution;
701
+ /** True if the message was injected by the system (e.g., auto-continue) and initiates a fresh run rather than continuing the current one. */
702
+ synthetic?: boolean;
703
+ /** True when the synthetic prompt was a deliberate operator action (`.`, `c` continue shortcut) rather than an automatic continuation — its timestamp is the turn's prompt time. */
704
+ userInitiated?: boolean;
701
705
  /** Provider-specific opaque payload used to reconstruct transport-native history. */
702
706
  providerPayload?: ProviderPayload;
703
707
  timestamp: number;
@@ -781,6 +785,8 @@ export interface AssistantMessage {
781
785
  timestamp: number;
782
786
  duration?: number;
783
787
  ttft?: number;
788
+ /** Local wall-clock time the response finished streaming (ms since epoch); stamped by the session at message_end so prompt→yield timing never depends on provider-reported duration. */
789
+ completedAt?: number;
784
790
  }
785
791
  export interface ToolResultMessage<TDetails = unknown> {
786
792
  role: "toolResult";
@@ -1,5 +1,16 @@
1
- import type { CredentialRankingStrategy, UsageProvider } from "../usage.js";
1
+ import type { CredentialRankingContext, CredentialRankingStrategy, UsageLimit, UsageProvider, UsageReport } from "../usage.js";
2
2
  export declare const antigravityUsageProvider: UsageProvider;
3
+ /** Map an Antigravity model id to its backend quota-counter key. */
4
+ export declare function getAntigravityCounterKeyForModel(modelId: string | undefined): string | undefined;
5
+ /**
6
+ * Scope an Antigravity report to the active model's backend counter, falling
7
+ * back to legacy default counters only when that backend has no limits.
8
+ *
9
+ * Exhaustion checks are only safe with a concrete backend counter. A no-model
10
+ * credential lookup (for example image-provider discovery) must not turn one
11
+ * exhausted family into a provider-wide block.
12
+ */
13
+ export declare function scopeAntigravityLimitsForModel(report: UsageReport, context: CredentialRankingContext | undefined): UsageLimit[];
3
14
  /**
4
15
  * Antigravity quotas are returned per backend counter (Anthropic / Google /
5
16
  * OpenAI) and can include both daily and weekly windows. `fetchAntigravityUsage`
@@ -1,5 +1,5 @@
1
1
  import type { FetchImpl, Provider } from "./types.js";
2
- export type UsageUnit = "percent" | "tokens" | "requests" | "usd" | "minutes" | "bytes" | "unknown";
2
+ export type UsageUnit = "percent" | "tokens" | "requests" | "credits" | "usd" | "minutes" | "bytes" | "unknown";
3
3
  export type UsageStatus = "ok" | "warning" | "exhausted" | "unknown";
4
4
  /** Time window for a limit (e.g. 5h, 7d, monthly). */
5
5
  export interface UsageWindow {
@@ -204,7 +204,7 @@ export interface ClientUsageClientSummary {
204
204
  export interface ClientUsageSummary {
205
205
  clients: ClientUsageClientSummary[];
206
206
  }
207
- export declare const usageUnitSchema: import("@oh-my-pi/omptype").FluentType<"bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd", "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd">;
207
+ export declare const usageUnitSchema: import("@oh-my-pi/omptype").FluentType<"bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd", "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd">;
208
208
  export declare const usageStatusSchema: import("@oh-my-pi/omptype").FluentType<"exhausted" | "ok" | "unknown" | "warning", "exhausted" | "ok" | "unknown" | "warning">;
209
209
  export declare const usageWindowSchema: import("@oh-my-pi/omptype").FluentType<{
210
210
  durationMs?: number | undefined;
@@ -223,14 +223,14 @@ export declare const usageAmountSchema: import("@oh-my-pi/omptype").FluentType<{
223
223
  limit?: number | undefined;
224
224
  remaining?: number | undefined;
225
225
  remainingFraction?: number | undefined;
226
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
226
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
227
227
  used?: number | undefined;
228
228
  usedFraction?: number | undefined;
229
229
  }, {
230
230
  limit?: number | undefined;
231
231
  remaining?: number | undefined;
232
232
  remainingFraction?: number | undefined;
233
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
233
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
234
234
  used?: number | undefined;
235
235
  usedFraction?: number | undefined;
236
236
  }>;
@@ -258,7 +258,7 @@ export declare const usageLimitSchema: import("@oh-my-pi/omptype").FluentType<{
258
258
  limit?: number | undefined;
259
259
  remaining?: number | undefined;
260
260
  remainingFraction?: number | undefined;
261
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
261
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
262
262
  used?: number | undefined;
263
263
  usedFraction?: number | undefined;
264
264
  };
@@ -288,7 +288,7 @@ export declare const usageLimitSchema: import("@oh-my-pi/omptype").FluentType<{
288
288
  limit?: number | undefined;
289
289
  remaining?: number | undefined;
290
290
  remainingFraction?: number | undefined;
291
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
291
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
292
292
  used?: number | undefined;
293
293
  usedFraction?: number | undefined;
294
294
  };
@@ -345,7 +345,7 @@ export declare const usageReportSchema: import("@oh-my-pi/omptype").FluentType<{
345
345
  limit?: number | undefined;
346
346
  remaining?: number | undefined;
347
347
  remainingFraction?: number | undefined;
348
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
348
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
349
349
  used?: number | undefined;
350
350
  usedFraction?: number | undefined;
351
351
  };
@@ -390,7 +390,7 @@ export declare const usageReportSchema: import("@oh-my-pi/omptype").FluentType<{
390
390
  limit?: number | undefined;
391
391
  remaining?: number | undefined;
392
392
  remainingFraction?: number | undefined;
393
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
393
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
394
394
  used?: number | undefined;
395
395
  usedFraction?: number | undefined;
396
396
  };
@@ -528,6 +528,10 @@ export interface CredentialRankingStrategy {
528
528
  primaryMs: number;
529
529
  secondaryMs: number;
530
530
  };
531
- /** Optional: priority boost for specific credential states (e.g., fresh 5h ticker start). */
532
- hasPriorityBoost?(primary: UsageLimit | undefined): boolean;
531
+ /**
532
+ * Optional: priority boost for specific credential states (e.g., fresh 5h
533
+ * ticker start). `primaryUncapped` is true only when the fetched report has
534
+ * an applicable secondary window but no applicable primary window.
535
+ */
536
+ hasPriorityBoost?(primary: UsageLimit | undefined, primaryUncapped?: boolean, context?: CredentialRankingContext): boolean;
533
537
  }
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-ai",
4
- "version": "18.0.7",
4
+ "version": "18.0.9",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Stencil Labs, Inc.",
@@ -37,10 +37,10 @@
37
37
  "fmt": "biome format --write ."
38
38
  },
39
39
  "dependencies": {
40
- "@oh-my-pi/omptype": "18.0.7",
41
- "@oh-my-pi/pi-catalog": "18.0.7",
42
- "@oh-my-pi/pi-utils": "18.0.7",
43
- "@oh-my-pi/pi-wire": "18.0.7"
40
+ "@oh-my-pi/omptype": "18.0.9",
41
+ "@oh-my-pi/pi-catalog": "18.0.9",
42
+ "@oh-my-pi/pi-utils": "18.0.9",
43
+ "@oh-my-pi/pi-wire": "18.0.9"
44
44
  },
45
45
  "devDependencies": {
46
46
  "@types/bun": "^1.3.14"
@@ -8,6 +8,7 @@ import { Database, type Statement } from "bun:sqlite";
8
8
  import * as fs from "node:fs/promises";
9
9
  import * as path from "node:path";
10
10
  import { parseAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
11
+ import { parseCloudflareAiGatewayCredential } from "@oh-my-pi/pi-catalog/wire/cloudflare-ai-gateway";
11
12
  import {
12
13
  getAgentDbPath,
13
14
  getDbBusyTimeoutMs,
@@ -215,10 +216,17 @@ function matchesReplacementCredential(
215
216
  if (incoming.type === "api_key") {
216
217
  if (existing.type !== "api_key") return false;
217
218
  if (existing.key === incoming.key) return true;
218
- if (provider !== "alibaba-token-plan") return false;
219
- const existingToken = parseAlibabaTokenPlanCredential(existing.key)?.token;
220
- const incomingToken = parseAlibabaTokenPlanCredential(incoming.key)?.token;
221
- return existingToken !== undefined && existingToken === incomingToken;
219
+ if (provider === "alibaba-token-plan") {
220
+ const existingToken = parseAlibabaTokenPlanCredential(existing.key)?.token;
221
+ const incomingToken = parseAlibabaTokenPlanCredential(incoming.key)?.token;
222
+ return existingToken !== undefined && existingToken === incomingToken;
223
+ }
224
+ if (provider === "cloudflare-ai-gateway") {
225
+ const existingToken = parseCloudflareAiGatewayCredential(existing.key)?.token;
226
+ const incomingToken = parseCloudflareAiGatewayCredential(incoming.key)?.token;
227
+ return existingToken !== undefined && existingToken === incomingToken;
228
+ }
229
+ return false;
222
230
  }
223
231
  const incomingIdentifiers = extractOAuthCredentialIdentifiers(incoming);
224
232
  const incomingIdentityKey = resolveProviderCredentialIdentityKey(provider, incomingIdentifiers);
@@ -208,7 +208,7 @@ const usageAmountSchema = type({
208
208
  "remaining?": "number",
209
209
  "usedFraction?": "number",
210
210
  "remainingFraction?": "number",
211
- unit: "'percent' | 'tokens' | 'requests' | 'usd' | 'minutes' | 'bytes' | 'unknown'",
211
+ unit: "'percent' | 'tokens' | 'requests' | 'credits' | 'usd' | 'minutes' | 'bytes' | 'unknown'",
212
212
  });
213
213
 
214
214
  const usageScopeSchema = type({
@@ -28,6 +28,7 @@ import type {
28
28
  OAuthProvider,
29
29
  OAuthProviderId,
30
30
  } from "./registry/oauth/types";
31
+ import { AUTHENTICATED_SENTINEL } from "./registry/types";
31
32
  import { getEnvApiKey, getEnvApiKeyName } from "./stream";
32
33
  import type { Provider } from "./types";
33
34
  import type {
@@ -1272,6 +1273,7 @@ type UsageRankedCandidate<T extends AuthCredential> = UsageCandidate<T> & {
1272
1273
  blocked: boolean;
1273
1274
  blockedUntil?: number;
1274
1275
  hasPriorityBoost: boolean;
1276
+ usageMeasured: boolean;
1275
1277
  planPriority: number;
1276
1278
  secondaryUsed: number;
1277
1279
  secondaryRequiredDrain: number;
@@ -2168,13 +2170,16 @@ export class AuthStorage {
2168
2170
  const windows = usage ? strategy.findWindowLimits(usage, args.rankingContext) : undefined;
2169
2171
  const primary = windows?.primary;
2170
2172
  const secondary = windows?.secondary;
2173
+ const usageMeasured = primary !== undefined || secondary !== undefined;
2174
+ const primaryUncapped = primary === undefined && secondary !== undefined;
2171
2175
  ranked.push({
2172
2176
  selection,
2173
2177
  usage,
2174
2178
  usageChecked,
2175
2179
  blocked,
2176
2180
  blockedUntil,
2177
- hasPriorityBoost: strategy.hasPriorityBoost?.(primary) ?? false,
2181
+ usageMeasured,
2182
+ hasPriorityBoost: strategy.hasPriorityBoost?.(primary, primaryUncapped, args.rankingContext) ?? false,
2178
2183
  planPriority: 0,
2179
2184
  secondaryUsed: this.#normalizeUsageFraction(secondary),
2180
2185
  secondaryRequiredDrain: this.#computeWindowRequiredDrain(
@@ -2748,6 +2753,34 @@ export class AuthStorage {
2748
2753
  return false;
2749
2754
  }
2750
2755
 
2756
+ /**
2757
+ * Like {@link hasAuth} but excludes providers whose only credential is the
2758
+ * self-resolving {@link AUTHENTICATED_SENTINEL} — the marker AWS/Vertex
2759
+ * transports return when a credential *source* merely exists (a stray
2760
+ * `~/.aws` profile, an EC2 instance role, Application Default Credentials)
2761
+ * without a usable key resolved yet. Default-model auto-selection uses this
2762
+ * so an ambiently-available provider (e.g. `amazon-bedrock` via an unrelated
2763
+ * AWS profile) does not win the startup default over a provider the user
2764
+ * actually signed into and then 403 on the first turn. Explicit selection
2765
+ * and picker visibility still go through {@link hasAuth}. See issue #9967.
2766
+ */
2767
+ hasConcreteAuth(provider: string): boolean {
2768
+ if (this.#runtimeOverrides.has(provider)) return true;
2769
+ if (this.#configOverrides.has(provider)) return true;
2770
+ if (this.#getCredentialsForProvider(provider).length > 0) return true;
2771
+ if ((provider === "amazon-bedrock" || provider === "bedrock-mantle") && $env.AWS_BEARER_TOKEN_BEDROCK?.trim()) {
2772
+ return true;
2773
+ }
2774
+ if (provider === "xai-oauth") {
2775
+ if ($env.XAI_OAUTH_TOKEN?.trim()) return true;
2776
+ } else {
2777
+ const envApiKey = getEnvApiKey(provider);
2778
+ if (envApiKey !== undefined && envApiKey !== AUTHENTICATED_SENTINEL) return true;
2779
+ }
2780
+ const fallback = this.#fallbackResolver?.(provider);
2781
+ return fallback !== undefined && fallback !== AUTHENTICATED_SENTINEL;
2782
+ }
2783
+
2751
2784
  /**
2752
2785
  * Whether a request could resolve a key for this provider, including
2753
2786
  * cross-provider env aliases (`xai-oauth` borrowing `XAI_API_KEY`).
@@ -4633,8 +4666,8 @@ export class AuthStorage {
4633
4666
  // scores are only comparable between measured windows, and the
4634
4667
  // clockless headroom fallback (0..1) must not let an account whose
4635
4668
  // usage fetch failed shadow a measured sibling.
4636
- const leftMeasured = left.usage !== null;
4637
- const rightMeasured = right.usage !== null;
4669
+ const leftMeasured = left.usageMeasured;
4670
+ const rightMeasured = right.usageMeasured;
4638
4671
  if (leftMeasured !== rightMeasured) return leftMeasured ? -1 : 1;
4639
4672
  // Required drain, descending: the account whose remaining quota must
4640
4673
  // burn fastest to avoid expiring unused at its reset comes first, so
@@ -4769,13 +4802,16 @@ export class AuthStorage {
4769
4802
  const windows = usage ? strategy.findWindowLimits(usage, args.rankingContext) : undefined;
4770
4803
  const primary = windows?.primary;
4771
4804
  const secondary = windows?.secondary;
4805
+ const usageMeasured = primary !== undefined || secondary !== undefined;
4806
+ const primaryUncapped = primary === undefined && secondary !== undefined;
4772
4807
  ranked.push({
4773
4808
  selection,
4774
4809
  usage,
4775
4810
  usageChecked,
4776
4811
  blocked,
4777
4812
  blockedUntil,
4778
- hasPriorityBoost: strategy.hasPriorityBoost?.(primary) ?? false,
4813
+ usageMeasured,
4814
+ hasPriorityBoost: strategy.hasPriorityBoost?.(primary, primaryUncapped, args.rankingContext) ?? false,
4779
4815
  planPriority: getOpenAICodexPlanPriority(usage, args.planRequirement),
4780
4816
  secondaryUsed: this.#normalizeUsageFraction(secondary),
4781
4817
  secondaryRequiredDrain: this.#computeWindowRequiredDrain(
@@ -1089,6 +1089,15 @@ function buildAdditionalModelRequestFields(
1089
1089
  };
1090
1090
  }
1091
1091
 
1092
+ if (mode === "effort") {
1093
+ // OpenAI-schema models on Bedrock (the GPT-5.x SKUs) reject the
1094
+ // Anthropic budget block with `unknown_parameter: 'thinking'` and take
1095
+ // `reasoning.effort` instead — same effort vocabulary the catalog
1096
+ // already bakes (low/medium/high/xhigh/max).
1097
+ const level = requireSupportedEffort(model, reasoning);
1098
+ return { reasoning: { effort: model.thinking?.effortMap?.[level] ?? level } };
1099
+ }
1100
+
1092
1101
  const level = requireSupportedEffort(model, reasoning);
1093
1102
  const defaultBudgets: Record<Effort, number> = {
1094
1103
  minimal: 1024,
@@ -4844,6 +4844,27 @@ export function buildCursorSystemPromptJsons(systemPrompt: readonly string[] | u
4844
4844
  return systemPrompts.map(content => JSON.stringify({ role: "system", content }));
4845
4845
  }
4846
4846
 
4847
+ function collectCursorToolHistory(messages: Message[], historyEnd: number) {
4848
+ const toolResults = new Map<string, ToolResultMessage>();
4849
+ const pairedToolCallIds = new Set<string>();
4850
+ for (let index = 0; index < historyEnd; index++) {
4851
+ const message = messages[index];
4852
+ if (message.role === "toolResult") {
4853
+ toolResults.set(message.toolCallId, message);
4854
+ } else if (message.role === "assistant") {
4855
+ for (const item of message.content) {
4856
+ if (item.type === "toolCall") pairedToolCallIds.add(item.id);
4857
+ }
4858
+ }
4859
+ }
4860
+ return { toolResults, pairedToolCallIds };
4861
+ }
4862
+
4863
+ function cursorOrphanToolResultText(result: ToolResultMessage): string {
4864
+ const prefix = result.isError ? "[Tool Error]" : "[Tool Result]";
4865
+ return `${prefix}\n${toolResultToText(result) || "(empty result)"}`;
4866
+ }
4867
+
4847
4868
  function buildRootPromptMessagesJson(
4848
4869
  messages: Message[],
4849
4870
  systemPromptIds: Uint8Array[],
@@ -4852,6 +4873,8 @@ function buildRootPromptMessagesJson(
4852
4873
  targetModelId?: string,
4853
4874
  ): Uint8Array[] {
4854
4875
  assertCursorKimiK3HistoryReplayable(messages, activeUserMessageIndex, targetModelId);
4876
+ const historyEnd = activeUserMessageIndex >= 0 ? activeUserMessageIndex : messages.length;
4877
+ const { pairedToolCallIds } = collectCursorToolHistory(messages, historyEnd);
4855
4878
  const entries: Uint8Array[] = [...systemPromptIds];
4856
4879
  const pushJson = (obj: unknown) => {
4857
4880
  const bytes = new TextEncoder().encode(JSON.stringify(obj));
@@ -4870,6 +4893,13 @@ function buildRootPromptMessagesJson(
4870
4893
  if (content.length === 0) continue;
4871
4894
  pushJson({ role: "assistant", content });
4872
4895
  } else if (msg.role === "toolResult") {
4896
+ if (!pairedToolCallIds.has(msg.toolCallId)) {
4897
+ pushJson({
4898
+ role: "assistant",
4899
+ content: [{ type: "text", text: cursorOrphanToolResultText(msg) }],
4900
+ });
4901
+ continue;
4902
+ }
4873
4903
  // Emit even when the result text is empty: the assistant `tool-call` is
4874
4904
  // already in history, so dropping the pair would replay an orphaned call.
4875
4905
  const toolCallId = normalizeToolCallId(msg.toolCallId);
@@ -5008,18 +5038,7 @@ function buildConversationTurns(
5008
5038
  ): Uint8Array[] {
5009
5039
  const turns: Uint8Array[] = [];
5010
5040
  const historyEnd = activeUserMessageIndex >= 0 ? activeUserMessageIndex : messages.length;
5011
- const toolResults = new Map<string, ToolResultMessage>();
5012
- const pairedToolCallIds = new Set<string>();
5013
- for (let index = 0; index < historyEnd; index++) {
5014
- const message = messages[index];
5015
- if (message.role === "toolResult") {
5016
- toolResults.set(message.toolCallId, message);
5017
- } else if (message.role === "assistant") {
5018
- for (const item of message.content) {
5019
- if (item.type === "toolCall") pairedToolCallIds.add(item.id);
5020
- }
5021
- }
5022
- }
5041
+ const { toolResults, pairedToolCallIds } = collectCursorToolHistory(messages, historyEnd);
5023
5042
 
5024
5043
  let i = 0;
5025
5044
  while (i < messages.length) {
@@ -5077,17 +5096,13 @@ function buildConversationTurns(
5077
5096
  stepBlobIds.push(storeCursorBlob(blobStore, toBinary(ConversationStepSchema, step)));
5078
5097
  }
5079
5098
  } else if (stepMsg.role === "toolResult" && !pairedToolCallIds.has(stepMsg.toolCallId)) {
5080
- const text = toolResultToText(stepMsg);
5081
- if (text) {
5082
- const prefix = stepMsg.isError ? "[Tool Error]" : "[Tool Result]";
5083
- const step = create(ConversationStepSchema, {
5084
- message: {
5085
- case: "assistantMessage",
5086
- value: create(AssistantMessageSchema, { text: `${prefix}\n${text}` }),
5087
- },
5088
- });
5089
- stepBlobIds.push(storeCursorBlob(blobStore, toBinary(ConversationStepSchema, step)));
5090
- }
5099
+ const step = create(ConversationStepSchema, {
5100
+ message: {
5101
+ case: "assistantMessage",
5102
+ value: create(AssistantMessageSchema, { text: cursorOrphanToolResultText(stepMsg) }),
5103
+ },
5104
+ });
5105
+ stepBlobIds.push(storeCursorBlob(blobStore, toBinary(ConversationStepSchema, step)));
5091
5106
  }
5092
5107
  i++;
5093
5108
  }
@@ -1,26 +1,130 @@
1
- import { createApiKeyLogin } from "./api-key-login";
2
- import type { OAuthLoginCallbacks } from "./oauth/types";
1
+ import { buildModel } from "@oh-my-pi/pi-catalog/build";
2
+ import {
3
+ CLOUDFLARE_AI_GATEWAY_ANTHROPIC_BASE_URL,
4
+ CLOUDFLARE_AI_GATEWAY_BASE_URL,
5
+ CLOUDFLARE_AI_GATEWAY_COMPAT_BASE_URL,
6
+ CLOUDFLARE_AI_GATEWAY_OPENAI_BASE_URL,
7
+ parseCloudflareAiGatewayCredential,
8
+ serializeCloudflareAiGatewayCredential,
9
+ } from "@oh-my-pi/pi-catalog/wire/cloudflare-ai-gateway";
10
+ import { $env } from "@oh-my-pi/pi-utils";
11
+ import * as AIError from "../error";
12
+ import { NO_AUTH_SENTINEL } from "../providers/openai-shared";
13
+ import type { OAuthController, OAuthLoginCallbacks } from "./oauth/types";
3
14
  import type { ProviderDefinition } from "./types";
4
15
 
5
16
  const AUTH_URL = "https://developers.cloudflare.com/ai-gateway/configuration/authentication/";
6
17
 
7
- /**
8
- * Login to Cloudflare AI Gateway.
9
- *
10
- * Opens browser to Cloudflare AI Gateway authentication docs and prompts for a gateway token/API key.
11
- * Returns the API key directly (not OAuthCredentials - this isn't OAuth).
12
- */
13
- export const loginCloudflareAiGateway = createApiKeyLogin({
14
- providerLabel: "Cloudflare AI Gateway",
15
- authUrl: AUTH_URL,
16
- instructions: "Copy your Cloudflare AI Gateway token/API key. Configure account/gateway base URL in models config.",
17
- promptMessage: "Paste your Cloudflare AI Gateway token/API key",
18
- placeholder: "cf-aig-...",
19
- validation: null,
20
- });
18
+ /** Collect the gateway credential used by CLI, setup-wizard, and TUI login callers. */
19
+ export async function loginCloudflareAiGateway(options: OAuthController): Promise<string> {
20
+ if (!options.onPrompt) {
21
+ throw new AIError.OnPromptRequiredError("Cloudflare AI Gateway");
22
+ }
23
+ options.onAuth?.({
24
+ url: AUTH_URL,
25
+ instructions: "Create an AI Gateway token with Run permission, then copy it here.",
26
+ });
27
+
28
+ const apiKey = await options.onPrompt({
29
+ message: "Paste your Cloudflare AI Gateway token/API key",
30
+ placeholder: "cfut_...",
31
+ });
32
+ if (options.signal?.aborted) throw new AIError.LoginCancelledError();
33
+ if (!apiKey.trim()) throw new AIError.ApiKeyRequiredError();
34
+
35
+ const accountId = await options.onPrompt({
36
+ message: "Enter your Cloudflare account ID",
37
+ placeholder: "32-character account ID",
38
+ });
39
+ if (options.signal?.aborted) throw new AIError.LoginCancelledError();
40
+ if (!accountId.trim()) throw new AIError.ConfigurationError("Cloudflare account ID is required");
41
+
42
+ const gatewayId = await options.onPrompt({
43
+ message: "Enter your Cloudflare AI Gateway ID",
44
+ placeholder: "default",
45
+ });
46
+ if (options.signal?.aborted) throw new AIError.LoginCancelledError();
47
+ if (!gatewayId.trim()) throw new AIError.ConfigurationError("Cloudflare AI Gateway ID is required");
48
+
49
+ return serializeCloudflareAiGatewayCredential(apiKey, accountId, gatewayId);
50
+ }
21
51
 
22
52
  export const cloudflareAiGatewayProvider = {
23
53
  id: "cloudflare-ai-gateway",
24
54
  name: "Cloudflare AI Gateway",
55
+ prepareModel: model => {
56
+ const hasGatewayPlaceholders = model.baseUrl.includes("<account>") || model.baseUrl.includes("<gateway>");
57
+ if (model.id.startsWith("anthropic/")) {
58
+ const requestModelId = model.id.slice("anthropic/".length).replaceAll(".", "-");
59
+ const baseUrl = hasGatewayPlaceholders ? CLOUDFLARE_AI_GATEWAY_ANTHROPIC_BASE_URL : model.baseUrl;
60
+ if (
61
+ model.api === "anthropic-messages" &&
62
+ model.baseUrl === baseUrl &&
63
+ model.requestModelId === requestModelId
64
+ ) {
65
+ return model;
66
+ }
67
+ return {
68
+ ...model,
69
+ api: "anthropic-messages",
70
+ baseUrl,
71
+ requestModelId,
72
+ };
73
+ }
74
+ if (model.id.startsWith("openai/")) {
75
+ const requestModelId = model.id.slice("openai/".length);
76
+ const baseUrl = hasGatewayPlaceholders ? CLOUDFLARE_AI_GATEWAY_OPENAI_BASE_URL : model.baseUrl;
77
+ if (model.api === "openai-responses" && model.baseUrl === baseUrl && model.requestModelId === requestModelId) {
78
+ return model;
79
+ }
80
+ return buildModel({
81
+ ...model,
82
+ api: "openai-responses",
83
+ baseUrl,
84
+ compat: model.compatConfig,
85
+ requestModelId,
86
+ });
87
+ }
88
+ if (model.id.startsWith("workers-ai/")) {
89
+ const baseUrl = hasGatewayPlaceholders ? CLOUDFLARE_AI_GATEWAY_COMPAT_BASE_URL : model.baseUrl;
90
+ if (model.api === "openai-completions" && model.baseUrl === baseUrl) {
91
+ return model;
92
+ }
93
+ return buildModel({
94
+ ...model,
95
+ api: "openai-completions",
96
+ baseUrl,
97
+ compat: model.compatConfig,
98
+ });
99
+ }
100
+ return model;
101
+ },
102
+ prepareRequest: (model, options) => {
103
+ const credential = parseCloudflareAiGatewayCredential(options.apiKey ?? $env.CLOUDFLARE_AI_GATEWAY_API_KEY ?? "");
104
+ if (!credential) return { model, options };
105
+ const accountId = credential.accountId ?? $env.CLOUDFLARE_ACCOUNT_ID;
106
+ const gatewayId = credential.gatewayId ?? $env.CLOUDFLARE_GATEWAY_ID;
107
+ let baseUrl = model.baseUrl;
108
+ if (baseUrl.startsWith(CLOUDFLARE_AI_GATEWAY_BASE_URL)) {
109
+ if (!accountId) throw new AIError.ConfigurationError("Cloudflare account ID is required");
110
+ if (!gatewayId) throw new AIError.ConfigurationError("Cloudflare AI Gateway ID is required");
111
+ baseUrl = baseUrl.replace("<account>", accountId).replace("<gateway>", gatewayId);
112
+ }
113
+
114
+ const isAnthropic = model.api === "anthropic-messages";
115
+ let headers = model.headers;
116
+ if (!isAnthropic) {
117
+ headers = { ...headers };
118
+ for (const name in headers) {
119
+ const normalized = name.toLowerCase();
120
+ if (normalized === "authorization" || normalized === "x-api-key") delete headers[name];
121
+ }
122
+ headers["cf-aig-authorization"] = `Bearer ${credential.token}`;
123
+ }
124
+ return {
125
+ model: { ...model, baseUrl, headers },
126
+ options: { ...options, apiKey: isAnthropic ? credential.token : NO_AUTH_SENTINEL },
127
+ };
128
+ },
25
129
  login: (cb: OAuthLoginCallbacks) => loginCloudflareAiGateway(cb),
26
130
  } as const satisfies ProviderDefinition;
@@ -303,10 +303,16 @@
303
303
  const message = document.getElementById("message");
304
304
 
305
305
  if (serverState.ok) {
306
+ const closeButton = document.querySelector(".btn");
306
307
  app.classList.add("success", "countdown");
307
308
  title.textContent = "Authentication Successful";
308
- message.innerHTML = "You have successfully logged in.<br>You can now close this tab.";
309
- setTimeout(() => window.close(), 3000);
309
+ message.textContent = "You have successfully logged in.";
310
+ window.close();
311
+ setTimeout(() => {
312
+ app.classList.remove("countdown");
313
+ closeButton.remove();
314
+ message.innerHTML = "You have successfully logged in.<br>Please close this tab manually.";
315
+ }, 300);
310
316
  } else {
311
317
  app.classList.add("error");
312
318
  title.textContent = "Authentication Failed";
@@ -29,6 +29,7 @@ export interface PreparedProviderRequest {
29
29
  }
30
30
 
31
31
  export type ProviderRequestPreparer = (model: Model<Api>, options: StreamOptions) => PreparedProviderRequest;
32
+ export type ProviderModelPreparer = (model: Model<Api>) => Model<Api>;
32
33
  export type ProviderSimpleOptionsMapper = (options: SimpleStreamOptions) => Readonly<Record<string, unknown>>;
33
34
 
34
35
  export interface ProviderModelDiscoveryConfig {
@@ -66,6 +67,8 @@ export interface ProviderDefinition {
66
67
  readonly envKeys?: KeyResolver;
67
68
  /** Provider transport can authenticate without a resolved API-key string. */
68
69
  readonly allowsMissingApiKey?: boolean;
70
+ /** Provider-owned model normalization that must run before API-specific option mapping. */
71
+ readonly prepareModel?: ProviderModelPreparer;
69
72
  /** Provider-owned request shaping applied before generic API dispatch. */
70
73
  readonly prepareRequest?: ProviderRequestPreparer;
71
74
  /** Provider-owned projection from the generic simple-stream option bag. */
package/src/stream.ts CHANGED
@@ -939,9 +939,10 @@ function streamDispatch<TApi extends Api>(
939
939
  return streamBedrock(model as Model<"bedrock-converse-stream">, context, requestOptions as BedrockOptions);
940
940
  }
941
941
 
942
- const prepareRequest = getProviderDefinition(model.provider)?.prepareRequest;
943
- const prepared = prepareRequest?.(model as Model<Api>, requestOptions as StreamOptions);
944
- const providerModel = prepared?.model ?? (model as Model<Api>);
942
+ const providerDefinition = getProviderDefinition(model.provider);
943
+ const requestModel = providerDefinition?.prepareModel?.(model) ?? model;
944
+ const prepared = providerDefinition?.prepareRequest?.(requestModel, requestOptions as StreamOptions);
945
+ const providerModel = prepared?.model ?? requestModel;
945
946
  const preparedOptions = prepared?.options ?? (requestOptions as StreamOptions);
946
947
  const apiKey = preparedOptions.apiKey || getEnvApiKey(providerModel.provider);
947
948
  if (!apiKey) {
@@ -1691,8 +1692,9 @@ function streamSimpleRequest<TApi extends Api>(
1691
1692
  ),
1692
1693
  );
1693
1694
  }
1694
- const providerOptions = mapOptionsForApi(model, requestOptions, apiKey);
1695
- return stream(model, context, providerOptions);
1695
+ const providerModel = getProviderDefinition(model.provider)?.prepareModel?.(model) ?? model;
1696
+ const providerOptions = mapOptionsForApi(providerModel, requestOptions, apiKey);
1697
+ return stream(providerModel, context, providerOptions);
1696
1698
  }
1697
1699
 
1698
1700
  export async function completeSimple<TApi extends Api>(
@@ -2080,8 +2082,8 @@ function mapOptionsForApi<TApi extends Api>(
2080
2082
  guardrailVersion: model.guardrailVersion ?? options?.guardrailVersion,
2081
2083
  guardrailTrace: model.guardrailTrace ?? options?.guardrailTrace,
2082
2084
  };
2083
- // Adaptive mode sends effort directly, no budget_tokens — skip budget inflation.
2084
- if (model.thinking?.mode === "anthropic-adaptive") {
2085
+ // Effort modes send effort directly, no budget_tokens — skip budget inflation.
2086
+ if (model.thinking?.mode === "effort" || model.thinking?.mode === "anthropic-adaptive") {
2085
2087
  return castApi<"bedrock-converse-stream">(bedrockBase);
2086
2088
  }
2087
2089
  const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
package/src/types.ts CHANGED
@@ -879,6 +879,10 @@ export interface DeveloperMessage {
879
879
  content: string | (TextContent | ImageContent)[];
880
880
  /** Who initiated this message for billing/attribution semantics. */
881
881
  attribution?: MessageAttribution;
882
+ /** True if the message was injected by the system (e.g., auto-continue) and initiates a fresh run rather than continuing the current one. */
883
+ synthetic?: boolean;
884
+ /** True when the synthetic prompt was a deliberate operator action (`.`, `c` continue shortcut) rather than an automatic continuation — its timestamp is the turn's prompt time. */
885
+ userInitiated?: boolean;
882
886
  /** Provider-specific opaque payload used to reconstruct transport-native history. */
883
887
  providerPayload?: ProviderPayload;
884
888
  timestamp: number; // Unix timestamp in milliseconds
@@ -976,6 +980,8 @@ export interface AssistantMessage {
976
980
  timestamp: number; // Unix timestamp in milliseconds
977
981
  duration?: number; // Request duration in milliseconds
978
982
  ttft?: number; // Time to first token in milliseconds
983
+ /** Local wall-clock time the response finished streaming (ms since epoch); stamped by the session at message_end so prompt→yield timing never depends on provider-reported duration. */
984
+ completedAt?: number;
979
985
  }
980
986
 
981
987
  export interface ToolResultMessage<TDetails = unknown> {
@@ -426,12 +426,19 @@ export const antigravityUsageProvider: UsageProvider = {
426
426
  supports: params => params.provider === "google-antigravity",
427
427
  };
428
428
 
429
- function getAntigravityCounterKeyForModel(context: CredentialRankingContext | undefined): string | undefined {
430
- const modelId = context?.modelId?.toLowerCase();
431
- if (!modelId) return undefined;
432
- if (modelId.startsWith("claude-")) return "anthropic";
433
- if (modelId.startsWith("gemini-") || modelId.startsWith("gemma-")) return "google";
434
- if (modelId.startsWith("gpt-") || modelId.startsWith("openai/")) return "openai";
429
+ /** Map an Antigravity model id to its backend quota-counter key. */
430
+ export function getAntigravityCounterKeyForModel(modelId: string | undefined): string | undefined {
431
+ const normalizedModelId = modelId?.toLowerCase();
432
+ if (!normalizedModelId) return undefined;
433
+ if (normalizedModelId.startsWith("claude-")) return "anthropic";
434
+ if (
435
+ normalizedModelId.startsWith("gemini-") ||
436
+ normalizedModelId.startsWith("gemma-") ||
437
+ normalizedModelId.startsWith("tab_")
438
+ ) {
439
+ return "google";
440
+ }
441
+ if (normalizedModelId.startsWith("gpt-") || normalizedModelId.startsWith("openai/")) return "openai";
435
442
  return undefined;
436
443
  }
437
444
 
@@ -440,14 +447,19 @@ function getAntigravityCounterLimits(report: UsageReport, counterKey: string): U
440
447
  return report.limits.filter(limit => limit.id.toLowerCase().startsWith(prefix));
441
448
  }
442
449
 
443
- // Exhaustion checks are only safe with a concrete backend counter. A no-model
444
- // Antigravity credential lookup (for example image-provider discovery) must
445
- // not turn one exhausted family into a provider-wide block.
446
- function scopeAntigravityLimitsForModel(
450
+ /**
451
+ * Scope an Antigravity report to the active model's backend counter, falling
452
+ * back to legacy default counters only when that backend has no limits.
453
+ *
454
+ * Exhaustion checks are only safe with a concrete backend counter. A no-model
455
+ * credential lookup (for example image-provider discovery) must not turn one
456
+ * exhausted family into a provider-wide block.
457
+ */
458
+ export function scopeAntigravityLimitsForModel(
447
459
  report: UsageReport,
448
460
  context: CredentialRankingContext | undefined,
449
461
  ): UsageLimit[] {
450
- const counterKey = getAntigravityCounterKeyForModel(context);
462
+ const counterKey = getAntigravityCounterKeyForModel(context?.modelId);
451
463
  if (!counterKey) return [];
452
464
  const backendLimits = getAntigravityCounterLimits(report, counterKey);
453
465
  if (backendLimits.length > 0) return backendLimits;
@@ -455,7 +467,7 @@ function scopeAntigravityLimitsForModel(
455
467
  }
456
468
 
457
469
  function rankAntigravityLimits(report: UsageReport, context: CredentialRankingContext | undefined): UsageLimit[] {
458
- const counterKey = getAntigravityCounterKeyForModel(context);
470
+ const counterKey = getAntigravityCounterKeyForModel(context?.modelId);
459
471
  if (!counterKey) return report.limits;
460
472
  return scopeAntigravityLimitsForModel(report, context);
461
473
  }
@@ -480,7 +492,7 @@ export const antigravityRankingStrategy: CredentialRankingStrategy = {
480
492
  // Always return a scope for Antigravity so missing/unknown model context
481
493
  // cannot fall through to AuthStorage's provider-wide block bucket.
482
494
  blockScope(context) {
483
- const counterKey = getAntigravityCounterKeyForModel(context);
495
+ const counterKey = getAntigravityCounterKeyForModel(context?.modelId);
484
496
  return `counter:${counterKey ?? "unknown"}`;
485
497
  },
486
498
  // Antigravity windows carry `durationMs` when the response identifies them
@@ -611,8 +611,11 @@ export const codexRankingStrategy: CredentialRankingStrategy = {
611
611
  return { primary: findLimit("primary"), secondary: findLimit("secondary") };
612
612
  },
613
613
  windowDefaults: { primaryMs: 60 * 60 * 1000, secondaryMs: 7 * 24 * 60 * 60 * 1000 },
614
- hasPriorityBoost(primary) {
615
- if (!primary) return false;
614
+ hasPriorityBoost(primary, primaryUncapped = false, context) {
615
+ // Chat plans can omit an uncapped primary window while retaining their
616
+ // weekly window. Spark always has a capped primary meter, so a missing
617
+ // Spark primary is incomplete rather than uncapped.
618
+ if (!primary) return primaryUncapped && !isCodexSparkRequest(context);
616
619
  const windowId = primary.scope.windowId?.toLowerCase();
617
620
  const durationMs = primary.window?.durationMs;
618
621
  const isFiveHourWindow =
package/src/usage/zai.ts CHANGED
@@ -51,6 +51,8 @@ interface ZaiQuotaPayload {
51
51
  msg?: string;
52
52
  data?: {
53
53
  limits?: ZaiUsageLimitItem[];
54
+ /** Coding-plan tier (e.g. "lite", "pro", "max") surfaced as the plan label. */
55
+ level?: string;
54
56
  };
55
57
  }
56
58
 
@@ -193,17 +195,33 @@ function buildModelUsageUrl(baseUrl: string, now: Date): string {
193
195
  }
194
196
 
195
197
  function getZaiCredentialLimits(report: UsageReport): UsageLimit[] {
196
- const limits = report.limits.filter(
197
- limit => limit.id.startsWith("zai:requests:") || limit.id.startsWith("zai:tokens:"),
198
+ return report.limits.filter(
199
+ limit =>
200
+ limit.id.startsWith("zai:requests:") ||
201
+ limit.id.startsWith("zai:tokens:") ||
202
+ limit.id.startsWith("zai:credits:"),
198
203
  );
199
- return limits;
204
+ }
205
+
206
+ function zaiLimitPressure(limit: UsageLimit): number {
207
+ const fraction = limit.amount.usedFraction;
208
+ return typeof fraction === "number" && Number.isFinite(fraction) ? fraction : -1;
200
209
  }
201
210
 
202
211
  function rankZaiRequestLimits(report: UsageReport): UsageLimit[] {
203
212
  const requestLimits = report.limits.filter(limit => limit.id.startsWith("zai:requests:"));
204
213
  const credentialLimits = getZaiCredentialLimits(report);
205
214
  const limits = requestLimits.length > 0 ? requestLimits : credentialLimits;
206
- const ranked = [...limits];
215
+ // Mixed-meter payloads (tokens + credits on the same plan) can repeat a
216
+ // window; keep the most-binding limit per window so a second 5h row never
217
+ // displaces the weekly window when primary/secondary are picked positionally.
218
+ const byWindow = new Map<number, UsageLimit>();
219
+ for (const limit of limits) {
220
+ const durationMs = limit.window?.durationMs ?? Number.POSITIVE_INFINITY;
221
+ const current = byWindow.get(durationMs);
222
+ if (!current || zaiLimitPressure(limit) > zaiLimitPressure(current)) byWindow.set(durationMs, limit);
223
+ }
224
+ const ranked = [...byWindow.values()];
207
225
  ranked.sort((left, right) => {
208
226
  const leftDuration = left.window?.durationMs ?? Number.POSITIVE_INFINITY;
209
227
  const rightDuration = right.window?.durationMs ?? Number.POSITIVE_INFINITY;
@@ -306,6 +324,33 @@ async function fetchZaiUsage(params: UsageFetchParams, ctx: UsageFetchContext):
306
324
  status: getUsageStatus(amount.usedFraction),
307
325
  });
308
326
  }
327
+ if (parsed.type === "CREDIT_LIMIT") {
328
+ // GLM Coding Plan windows (e.g. 12k credits / 5h + 60k credits / week):
329
+ // `usage` is the plan's credit allotment, `currentValue` the spend.
330
+ // `percentage` is a server-rounded integer (11 for 1438/12000 ≈ 11.98%),
331
+ // so prefer the exact ratio and fall back to it only without absolutes.
332
+ const window = buildZaiWindow(parsed);
333
+ const hasAbsoluteMeter = parsed.currentValue !== undefined && parsed.usage !== undefined && parsed.usage > 0;
334
+ const amount = buildUsageAmount({
335
+ used: parsed.currentValue,
336
+ limit: parsed.usage,
337
+ remaining: parsed.remaining,
338
+ percentage: hasAbsoluteMeter ? undefined : parsed.percentage,
339
+ unit: "credits",
340
+ });
341
+ limits.push({
342
+ id: `zai:credits:${window.id}`,
343
+ label: `ZAI ${window.label} Credit Quota`,
344
+ scope: {
345
+ provider: params.provider,
346
+ windowId: window.id,
347
+ shared: true,
348
+ },
349
+ window,
350
+ amount,
351
+ status: getUsageStatus(amount.usedFraction),
352
+ });
353
+ }
309
354
  }
310
355
 
311
356
  if (limits.length === 0) return null;
@@ -318,6 +363,7 @@ async function fetchZaiUsage(params: UsageFetchParams, ctx: UsageFetchContext):
318
363
  endpoint: url,
319
364
  accountId: credential.accountId,
320
365
  email: credential.email,
366
+ ...(typeof payload.data?.level === "string" && payload.data.level ? { planType: payload.data.level } : {}),
321
367
  },
322
368
  raw: payload,
323
369
  };
package/src/usage.ts CHANGED
@@ -6,7 +6,7 @@
6
6
  */
7
7
  import { type } from "@oh-my-pi/omptype";
8
8
  import type { FetchImpl, Provider } from "./types";
9
- export type UsageUnit = "percent" | "tokens" | "requests" | "usd" | "minutes" | "bytes" | "unknown";
9
+ export type UsageUnit = "percent" | "tokens" | "requests" | "credits" | "usd" | "minutes" | "bytes" | "unknown";
10
10
 
11
11
  export type UsageStatus = "ok" | "warning" | "exhausted" | "unknown";
12
12
 
@@ -240,7 +240,9 @@ export interface ClientUsageSummary {
240
240
 
241
241
  // ─── Zod schemas (wire-shape validation for the broker `/v1/usage` endpoint) ─
242
242
 
243
- export const usageUnitSchema = type("'percent' | 'tokens' | 'requests' | 'usd' | 'minutes' | 'bytes' | 'unknown'");
243
+ export const usageUnitSchema = type(
244
+ "'percent' | 'tokens' | 'requests' | 'credits' | 'usd' | 'minutes' | 'bytes' | 'unknown'",
245
+ );
244
246
  export const usageStatusSchema = type("'ok' | 'warning' | 'exhausted' | 'unknown'");
245
247
 
246
248
  export const usageWindowSchema = type({
@@ -412,6 +414,14 @@ export interface CredentialRankingStrategy {
412
414
  primaryMs: number;
413
415
  secondaryMs: number;
414
416
  };
415
- /** Optional: priority boost for specific credential states (e.g., fresh 5h ticker start). */
416
- hasPriorityBoost?(primary: UsageLimit | undefined): boolean;
417
+ /**
418
+ * Optional: priority boost for specific credential states (e.g., fresh 5h
419
+ * ticker start). `primaryUncapped` is true only when the fetched report has
420
+ * an applicable secondary window but no applicable primary window.
421
+ */
422
+ hasPriorityBoost?(
423
+ primary: UsageLimit | undefined,
424
+ primaryUncapped?: boolean,
425
+ context?: CredentialRankingContext,
426
+ ): boolean;
417
427
  }