@oh-my-pi/pi-ai 18.2.0 → 18.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/CHANGELOG.md +53 -0
  2. package/README.md +2 -0
  3. package/dist/types/auth/sqlite-credential-store.d.ts +2 -1
  4. package/dist/types/auth-broker/remote-store.d.ts +17 -0
  5. package/dist/types/auth-gateway/index.d.ts +1 -0
  6. package/dist/types/auth-gateway/session-state.d.ts +118 -0
  7. package/dist/types/auth-storage.d.ts +17 -0
  8. package/dist/types/error/body-error.d.ts +15 -0
  9. package/dist/types/error/flags.d.ts +16 -0
  10. package/dist/types/error/index.d.ts +1 -0
  11. package/dist/types/index.d.ts +1 -0
  12. package/dist/types/oneshot-retry.d.ts +6 -0
  13. package/dist/types/provider-session-state.d.ts +46 -0
  14. package/dist/types/providers/amazon-bedrock.d.ts +3 -0
  15. package/dist/types/providers/aws-sigv4.d.ts +12 -0
  16. package/dist/types/providers/openai-codex/request-transformer.d.ts +27 -0
  17. package/dist/types/providers/openai-responses.d.ts +15 -0
  18. package/dist/types/providers/openai-shared.d.ts +20 -3
  19. package/dist/types/registry/oauth/perplexity.d.ts +1 -7
  20. package/dist/types/registry/oauth/types.d.ts +8 -0
  21. package/dist/types/stream.d.ts +2 -0
  22. package/dist/types/types.d.ts +3 -1
  23. package/dist/types/usage/openai-codex.d.ts +3 -1
  24. package/dist/types/usage.d.ts +11 -1
  25. package/dist/types/utils/block-symbols.d.ts +36 -0
  26. package/dist/types/utils/openai-http.d.ts +2 -0
  27. package/dist/types/utils/retry-after.d.ts +2 -0
  28. package/dist/types/utils/schema/wire.d.ts +4 -5
  29. package/dist/types/utils.d.ts +9 -0
  30. package/package.json +6 -6
  31. package/src/auth/sqlite-credential-store.ts +8 -33
  32. package/src/auth-broker/remote-store.ts +73 -8
  33. package/src/auth-broker/wire-schemas.ts +1 -0
  34. package/src/auth-gateway/index.ts +1 -0
  35. package/src/auth-gateway/server.ts +186 -74
  36. package/src/auth-gateway/session-state.ts +312 -0
  37. package/src/auth-storage.ts +146 -15
  38. package/src/error/body-error.ts +310 -0
  39. package/src/error/flags.ts +63 -13
  40. package/src/error/index.ts +1 -0
  41. package/src/error/retryable.ts +2 -0
  42. package/src/index.ts +1 -0
  43. package/src/oneshot-retry.ts +13 -3
  44. package/src/provider-session-state.ts +56 -0
  45. package/src/providers/amazon-bedrock.ts +20 -3
  46. package/src/providers/anthropic-messages-server.ts +104 -23
  47. package/src/providers/anthropic-signature.ts +5 -2
  48. package/src/providers/anthropic.ts +101 -15
  49. package/src/providers/aws-sigv4.ts +16 -5
  50. package/src/providers/cursor.ts +60 -10
  51. package/src/providers/devin.ts +82 -28
  52. package/src/providers/openai-chat-server.ts +4 -0
  53. package/src/providers/openai-codex/request-transformer.ts +36 -0
  54. package/src/providers/openai-codex-responses.ts +35 -12
  55. package/src/providers/openai-completions.ts +49 -12
  56. package/src/providers/openai-reasoning-fallback.ts +6 -6
  57. package/src/providers/openai-responses-server.ts +2 -1
  58. package/src/providers/openai-responses.ts +52 -4
  59. package/src/providers/openai-shared.ts +199 -51
  60. package/src/registry/oauth/perplexity.ts +94 -28
  61. package/src/registry/oauth/types.ts +9 -0
  62. package/src/stream.ts +23 -2
  63. package/src/types.ts +3 -0
  64. package/src/usage/claude.ts +33 -0
  65. package/src/usage/google-antigravity.ts +8 -2
  66. package/src/usage/openai-codex.ts +94 -11
  67. package/src/usage.ts +8 -1
  68. package/src/utils/block-symbols.ts +57 -0
  69. package/src/utils/http-inspector.ts +20 -0
  70. package/src/utils/openai-http.ts +39 -3
  71. package/src/utils/retry-after.ts +12 -0
  72. package/src/utils/schema/normalize.ts +3 -3
  73. package/src/utils/schema/stamps.ts +33 -45
  74. package/src/utils/schema/wire.ts +9 -7
  75. package/src/utils.ts +67 -22
@@ -1,15 +1,6 @@
1
1
  /**
2
- * Perplexity login and token refresh.
3
- *
4
- * Login paths (in priority order):
5
- * 1. macOS native app: reads JWT from NSUserDefaults (`defaults read ai.perplexity.mac authToken`)
6
- * 2. HTTP email OTP: `GET /api/auth/csrf` → `POST /api/auth/signin-email` → `POST /api/auth/signin-otp`
7
- *
8
- * No browser or manual cookie paste required.
9
- * Refresh: Socket.IO `refreshJWT` RPC over authenticated WebSocket connection.
10
- *
11
- * Protocol: Engine.IO v4 + Socket.IO v4 over WebSocket (bypasses Cloudflare managed challenge).
12
- * Architecture reverse-engineered from Perplexity macOS app (ai.perplexity.mac).
2
+ * Perplexity login via legacy macOS session borrowing, host-managed browser SSO,
3
+ * or HTTP email OTP (including authenticator challenges).
13
4
  */
14
5
  import * as os from "node:os";
15
6
  import { $env } from "@oh-my-pi/pi-utils";
@@ -80,10 +71,7 @@ function jwtToCredentials(jwt: string, email?: string): OAuthCredentials {
80
71
  // Desktop app extraction
81
72
  // ---------------------------------------------------------------------------
82
73
 
83
- /**
84
- * Read the Perplexity JWT from the native macOS Catalyst app's UserDefaults.
85
- * Tokens are stored in NSUserDefaults (not Keychain), readable by any same-UID process.
86
- */
74
+ /** Read the legacy ai.perplexity.mac app's session; newer Mac apps use a restricted Keychain. */
87
75
  async function extractFromNativeApp(): Promise<string | null> {
88
76
  if (os.platform() !== "darwin") return null;
89
77
 
@@ -99,7 +87,7 @@ async function extractFromNativeApp(): Promise<string | null> {
99
87
  }
100
88
 
101
89
  // ---------------------------------------------------------------------------
102
- // Socket.IO email OTP login
90
+ // HTTP email OTP login
103
91
  // ---------------------------------------------------------------------------
104
92
 
105
93
  /**
@@ -274,32 +262,110 @@ async function httpEmailLogin(ctrl: OAuthController): Promise<OAuthCredentials>
274
262
  return jwtToCredentials(token, trimmedEmail);
275
263
  }
276
264
 
265
+ // ---------------------------------------------------------------------------
266
+ // Browser SSO login
267
+ // ---------------------------------------------------------------------------
268
+
269
+ const SESSION_COOKIE_NAME = "__Secure-next-auth.session-token";
270
+ const PERPLEXITY_BASE_URL = "https://www.perplexity.ai";
271
+
272
+ async function browserSsoLogin(ctrl: OAuthController): Promise<OAuthCredentials> {
273
+ if (!ctrl.onBrowserSession) {
274
+ throw new AIError.OAuthError("Browser SSO is unavailable in this client", {
275
+ kind: "validation",
276
+ provider: "perplexity",
277
+ });
278
+ }
279
+ ctrl.onProgress?.("Complete Perplexity sign-in in the browser window. Choose SSO for your organization.");
280
+ const token = (
281
+ await ctrl.onBrowserSession(
282
+ {
283
+ url: `${PERPLEXITY_BASE_URL}/auth/signin`,
284
+ cookieNames: [SESSION_COOKIE_NAME, "next-auth.session-token"],
285
+ },
286
+ ctrl.signal,
287
+ )
288
+ ).trim();
289
+ if (ctrl.signal?.aborted) throw new AIError.LoginCancelledError();
290
+ if (!token || /[\s;]/.test(token)) {
291
+ throw new AIError.OAuthError("Perplexity SSO captured an invalid session cookie", {
292
+ kind: "validation",
293
+ provider: "perplexity",
294
+ });
295
+ }
296
+
297
+ ctrl.onProgress?.("Validating Perplexity session...");
298
+ const response = await (ctrl.fetch ?? fetch)(`${PERPLEXITY_BASE_URL}/api/auth/session`, {
299
+ headers: {
300
+ Cookie: `${SESSION_COOKIE_NAME}=${token}`,
301
+ "User-Agent": APP_USER_AGENT,
302
+ "X-App-ApiVersion": API_VERSION,
303
+ },
304
+ redirect: "error",
305
+ signal: ctrl.signal,
306
+ });
307
+ if (!response.ok) {
308
+ throw new AIError.ProviderHttpError(`Perplexity session validation failed (${response.status})`, response.status);
309
+ }
310
+ let email: unknown;
311
+ try {
312
+ const session = (await response.json()) as { user?: { email?: unknown } } | null;
313
+ email = session?.user?.email;
314
+ } catch (error) {
315
+ if (ctrl.signal?.aborted) throw new AIError.LoginCancelledError();
316
+ if (!(error instanceof SyntaxError)) throw error;
317
+ throw new AIError.OAuthError("Perplexity returned an invalid session response", {
318
+ kind: "validation",
319
+ provider: "perplexity",
320
+ });
321
+ }
322
+ if (ctrl.signal?.aborted) throw new AIError.LoginCancelledError();
323
+ if (typeof email !== "string" || !email.trim()) {
324
+ throw new AIError.OAuthError("Perplexity session is invalid or expired. Sign in again.", {
325
+ kind: "validation",
326
+ provider: "perplexity",
327
+ });
328
+ }
329
+ return jwtToCredentials(token, email.trim());
330
+ }
331
+
277
332
  // ---------------------------------------------------------------------------
278
333
  // Public API
279
334
  // ---------------------------------------------------------------------------
280
335
 
281
- /**
282
- * Login to Perplexity.
283
- *
284
- * Tries auto-extraction from the desktop app, then runs HTTP email OTP login.
285
- *
286
- * No browser/manual token paste fallback is used.
287
- */
336
+ /** Prefer legacy app borrowing, then offer browser SSO when the host supports it. */
288
337
  export async function loginPerplexity(ctrl: OAuthController): Promise<OAuthCredentials> {
289
- if (!ctrl.onPrompt) {
290
- throw new AIError.OnPromptRequiredError("Perplexity");
291
- }
338
+ if (!ctrl.onPrompt) throw new AIError.OnPromptRequiredError("Perplexity");
339
+ if (ctrl.signal?.aborted) throw new AIError.LoginCancelledError();
292
340
 
293
- // Path 1: Native macOS app JWT (skip if PI_AUTH_NO_BORROW=1)
294
341
  if (!$env.PI_AUTH_NO_BORROW) {
295
342
  ctrl.onProgress?.("Checking for Perplexity desktop app...");
296
343
  const nativeJwt = await extractFromNativeApp();
344
+ if (ctrl.signal?.aborted) throw new AIError.LoginCancelledError();
297
345
  if (nativeJwt) {
298
346
  ctrl.onProgress?.("Found Perplexity JWT from native app");
299
347
  return jwtToCredentials(nativeJwt);
300
348
  }
301
349
  }
302
350
 
303
- // Path 2: HTTP email OTP
351
+ if (ctrl.onBrowserSession) {
352
+ const method = (
353
+ await ctrl.onPrompt({
354
+ message: "Login method: sso (browser) or email; blank for sso",
355
+ placeholder: "sso / email",
356
+ allowEmpty: true,
357
+ })
358
+ )
359
+ .trim()
360
+ .toLowerCase();
361
+ if (ctrl.signal?.aborted) throw new AIError.LoginCancelledError();
362
+ if (!method || method === "sso") return browserSsoLogin(ctrl);
363
+ if (method !== "email") {
364
+ throw new AIError.OAuthError("Choose sso or email for Perplexity login", {
365
+ kind: "validation",
366
+ provider: "perplexity",
367
+ });
368
+ }
369
+ }
304
370
  return httpEmailLogin(ctrl);
305
371
  }
@@ -70,12 +70,21 @@ export interface OAuthProviderInfo {
70
70
  storeCredentialsAs?: string;
71
71
  }
72
72
 
73
+ /** Sign-in URL and accepted cookies for an isolated, host-owned browser. */
74
+ export type OAuthBrowserSessionRequest = {
75
+ url: string;
76
+ /** Cookie names in preference order; return the first non-empty matching value. */
77
+ cookieNames: readonly string[];
78
+ };
79
+
73
80
  export interface OAuthController {
74
81
  onAuth?(info: OAuthAuthInfo): void;
75
82
  onProgress?(message: string): void;
76
83
  /** Request pasted callback input; stop any visible prompt when `signal` aborts. */
77
84
  onManualCodeInput?(signal?: AbortSignal): Promise<string>;
78
85
  onPrompt?(prompt: OAuthPrompt): Promise<string>;
86
+ /** Complete browser login and return one matching cookie value privately. Reject on cancellation or failure. */
87
+ onBrowserSession?(request: OAuthBrowserSessionRequest, signal?: AbortSignal): Promise<string>;
79
88
  signal?: AbortSignal;
80
89
  fetch?: FetchImpl;
81
90
  }
package/src/stream.ts CHANGED
@@ -193,6 +193,8 @@ let providerInFlightHeartbeatWriterOverride:
193
193
  | undefined;
194
194
  let providerInFlightLeaseRemoverOverride: ((leasePath: string) => Promise<void>) | undefined;
195
195
  let providerInFlightWaitObserverOverride: ((provider: string) => void) | undefined;
196
+ let providerInFlightLockCreatedObserverOverride: ((lockDir: string) => Promise<void>) | undefined;
197
+ let providerInFlightLockIdentifiedObserverOverride: ((lockDir: string) => Promise<void>) | undefined;
196
198
 
197
199
  export function configureProviderMaxInFlightRequests(limits: Record<string, number> | undefined): void {
198
200
  configuredProviderMaxInFlightRequests = limits ?? {};
@@ -365,12 +367,21 @@ async function acquireProviderInFlightLock(provider: string, signal?: AbortSigna
365
367
  if (signal?.aborted) throw signal.reason ?? new AIError.AbortError("Provider request aborted before dispatch");
366
368
  try {
367
369
  await fs.mkdir(lockDir);
368
- const lockIdentity = await readProviderInFlightLockIdentity(lockDir);
370
+ await providerInFlightLockCreatedObserverOverride?.(lockDir);
371
+ let lockIdentity: ProviderInFlightLockIdentity;
372
+ try {
373
+ lockIdentity = await readProviderInFlightLockIdentity(lockDir);
374
+ } catch (error) {
375
+ if (isEnoent(error)) continue;
376
+ throw error;
377
+ }
369
378
  const token = crypto.randomUUID();
370
379
  try {
380
+ await providerInFlightLockIdentifiedObserverOverride?.(lockDir);
371
381
  await writeProviderInFlightInfo(lockDir, token);
372
382
  } catch (error) {
373
383
  await releaseProviderInFlightLockDirIfSame(lockDir, lockIdentity);
384
+ if (isEnoent(error)) continue;
374
385
  throw error;
375
386
  }
376
387
  return async () => {
@@ -623,6 +634,12 @@ export const __providerInFlightForTesting = {
623
634
  setWaitObserver(observer: ((provider: string) => void) | undefined): void {
624
635
  providerInFlightWaitObserverOverride = observer;
625
636
  },
637
+ setLockCreatedObserver(observer: ((lockDir: string) => Promise<void>) | undefined): void {
638
+ providerInFlightLockCreatedObserverOverride = observer;
639
+ },
640
+ setLockIdentifiedObserver(observer: ((lockDir: string) => Promise<void>) | undefined): void {
641
+ providerInFlightLockIdentifiedObserverOverride = observer;
642
+ },
626
643
  providerDir(provider: string): string {
627
644
  return providerInFlightDir(provider);
628
645
  },
@@ -1061,7 +1078,11 @@ async function resolveWithThinkingLoopRetries(
1061
1078
  onAttempt?: (message: AssistantMessage) => void,
1062
1079
  ): Promise<AssistantMessage> {
1063
1080
  const dispatchAttempt = async (): Promise<AssistantMessage> => {
1064
- const message = await dispatch().result();
1081
+ const response = dispatch();
1082
+ for await (const _event of response) {
1083
+ // Completion callers do not consume deltas; drain them as they arrive to avoid retaining the response history.
1084
+ }
1085
+ const message = await response.result();
1065
1086
  onAttempt?.(message);
1066
1087
  return message;
1067
1088
  };
package/src/types.ts CHANGED
@@ -1055,6 +1055,8 @@ export interface AssistantMessage {
1055
1055
  errorMessage?: string;
1056
1056
  /** Stable recovery-classification text when errorMessage includes display-only diagnostics. */
1057
1057
  errorClassificationMessage?: string;
1058
+ /** True only when an exact request-body-read timeout failed on a full Responses replay, not a previous-response delta. */
1059
+ requestBodyReadTimeoutFullReplay?: boolean;
1058
1060
  /** Per-tool abort messages used when an aborted assistant turn needs different placeholder results per tool call. */
1059
1061
  toolCallAbortMessages?: Record<string, string>;
1060
1062
  /** HTTP status surfaced by the provider when the request failed. Populated by every provider's catch block alongside `errorMessage` so consumers (auth retry, telemetry, UI) can branch without regex-scraping the message. */
@@ -1187,6 +1189,7 @@ export type CursorTodoSyncHandler = (
1187
1189
  snapshot: CursorTodoSnapshot | null,
1188
1190
  toolCallId: string,
1189
1191
  error: string | null,
1192
+ origin?: "read" | "update",
1190
1193
  ) => ToolResultMessage;
1191
1194
 
1192
1195
  export interface CursorShellStreamCallbacks {
@@ -22,6 +22,8 @@ import { HOUR_MS, parseIsoTimestamp, WEEK_MS } from "./shared";
22
22
  const DEFAULT_ENDPOINT = "https://api.anthropic.com/api/oauth";
23
23
  const MAX_ATTEMPTS = 3;
24
24
  const BASE_RETRY_DELAY_MS = 500;
25
+ /** Shared windows that gate every Claude request, whatever the model. */
26
+ const CLAUDE_SHARED_GATE_WINDOW_IDS = ["5h", "7d"] as const;
25
27
 
26
28
  const CLAUDE_HEADERS = {
27
29
  accept: "application/json, text/plain, */*",
@@ -921,5 +923,36 @@ export const claudeRankingStrategy: CredentialRankingStrategy = {
921
923
  const kind = getClaudeModelKind(context);
922
924
  return kind === "fable" || kind === "mythos" ? `tier:${kind}` : undefined;
923
925
  },
926
+ /**
927
+ * A reactive Fable/Mythos block carries the reset the 429 reported, but
928
+ * Anthropic can restore the tier earlier (plan change, corrected counter),
929
+ * and the block then idles a usable account for days. Judge each tier scope
930
+ * against the limits that actually gate a request of that kind — its own
931
+ * weekly row plus the shared umbrella windows — so a healthy report lifts
932
+ * the block while a spent shared 5-hour wall keeps it.
933
+ *
934
+ * Only Fable/Mythos appear: {@link blockScope} scopes reactive blocks for
935
+ * those tiers alone, so no other scope can exist to heal.
936
+ */
937
+ healableBlockScopes(report) {
938
+ const sharedLimits = report.limits.filter(limit => limit.scope.shared === true);
939
+ // The endpoint returns a report as soon as one window parses, and a tier
940
+ // 429 can be caused by a shared wall. A payload missing a shared gate
941
+ // leaves the block's cause unknown, so vouch for nothing rather than
942
+ // clear a block that still holds.
943
+ const everySharedGateReported = CLAUDE_SHARED_GATE_WINDOW_IDS.every(windowId =>
944
+ sharedLimits.some(limit => limit.scope.windowId === windowId || limit.window?.id === windowId),
945
+ );
946
+ if (!everySharedGateReported) return [];
947
+ const tiers = new Set<string>();
948
+ for (const limit of report.limits) {
949
+ const tier = limit.scope.tier;
950
+ if (tier === "fable" || tier === "mythos") tiers.add(tier);
951
+ }
952
+ return [...tiers].map(tier => ({
953
+ blockScope: `tier:${tier}`,
954
+ limits: [...sharedLimits, ...report.limits.filter(limit => limit.scope.tier === tier)],
955
+ }));
956
+ },
924
957
  windowDefaults: { primaryMs: 5 * 60 * 60 * 1000, secondaryMs: 7 * 24 * 60 * 60 * 1000 },
925
958
  };
@@ -355,18 +355,24 @@ function buildQuotaSummaryReport(
355
355
  );
356
356
  const amount = buildQuotaSummaryAmount(bucket);
357
357
  const counterKeys = getQuotaSummaryCounterKeys(group, bucket);
358
+ const sharedGroup =
359
+ counterKeys.length > 1
360
+ ? `${bucket.bucketId ?? group?.displayName ?? "third-party"}:${window?.id ?? bucket.window ?? "default"}`
361
+ : undefined;
358
362
  for (const counterKey of counterKeys) {
359
363
  const counterName = getQuotaSummaryCounterName(counterKey);
360
364
  const windowId = window?.id ?? bucket.window ?? bucket.bucketId ?? "default";
365
+ const label =
366
+ sharedGroup !== undefined ? "Claude & GPT (shared)" : counterKey === "google" ? "Gemini" : counterName;
361
367
  limits.push({
362
368
  id: `${params.provider}:${counterKey}:default:${bucket.bucketId ?? windowId}`,
363
- label: counterName ? `Usage (${counterName})` : (group?.displayName ?? bucket.displayName ?? "Usage"),
369
+ label: label ?? group?.displayName ?? bucket.displayName ?? "Usage",
364
370
  scope: {
365
371
  provider: params.provider,
366
372
  accountId: params.credential.accountId,
367
373
  projectId: params.credential.projectId,
368
374
  windowId,
369
- ...(counterKeys.length > 1 ? { shared: true } : {}),
375
+ ...(sharedGroup !== undefined ? { shared: true, sharedGroup } : {}),
370
376
  },
371
377
  window,
372
378
  amount,
@@ -43,10 +43,23 @@ interface CodexUsageAdditionalRateLimitPayload {
43
43
  rate_limit?: CodexUsageRateLimitPayload | null;
44
44
  }
45
45
 
46
+ interface CodexUsageCreditsPayload {
47
+ has_credits?: boolean;
48
+ unlimited?: boolean;
49
+ overage_limit_reached?: boolean;
50
+ balance?: string | number;
51
+ }
52
+
53
+ interface CodexUsageSpendControlPayload {
54
+ reached?: boolean;
55
+ }
56
+
46
57
  interface CodexUsagePayload {
47
58
  plan_type?: string;
48
59
  rate_limit?: CodexUsageRateLimitPayload | null;
49
60
  additional_rate_limits?: CodexUsageAdditionalRateLimitPayload[] | null;
61
+ credits?: CodexUsageCreditsPayload | null;
62
+ spend_control?: CodexUsageSpendControlPayload | null;
50
63
  }
51
64
 
52
65
  interface ParsedUsageWindow {
@@ -72,6 +85,13 @@ interface ParsedUsage {
72
85
  primary?: ParsedUsageWindow;
73
86
  secondary?: ParsedUsageWindow;
74
87
  additional: ParsedAdditionalUsage[];
88
+ /**
89
+ * True when the account can still serve requests on credits after its plan
90
+ * windows report `limit_reached`. `/wham/usage` only describes the *plan*
91
+ * allowance, so without this a credit-funded account looks permanently
92
+ * exhausted until the weekly reset while `/responses` keeps accepting it.
93
+ */
94
+ creditOverage: boolean;
75
95
  raw: CodexUsagePayload;
76
96
  }
77
97
 
@@ -161,6 +181,28 @@ function parseAdditionalRateLimit(payload: unknown): ParsedAdditionalUsage | nul
161
181
  return { limitName, meteredFeature, allowed, limitReached, primary, secondary };
162
182
  }
163
183
 
184
+ /**
185
+ * True when paid credits can still fund plan-window overage. Codex CLI never
186
+ * gates on `/wham/usage`, so once the plan allowance is spent it keeps working
187
+ * off this balance; omp must mirror that or it parks a perfectly usable account
188
+ * until the weekly reset.
189
+ *
190
+ * Scoped to the plan verdict on purpose. `credits` describes the account's
191
+ * overage funding for the plan windows, and nothing in the payload says a
192
+ * balance covers a separate metered feature (Spark, reserve). A denial that is
193
+ * not plan exhaustion is left alone for the same reason: credits answer
194
+ * "allowance spent", not "request refused".
195
+ */
196
+ function hasPlanCreditOverage(payload: Record<string, unknown>, planLimitReached: boolean | undefined): boolean {
197
+ if (planLimitReached !== true) return false;
198
+ const credits = isRecord(payload.credits) ? payload.credits : undefined;
199
+ if (!credits) return false;
200
+ if (credits.unlimited !== true && credits.has_credits !== true) return false;
201
+ if (credits.overage_limit_reached === true) return false;
202
+ const spendControl = isRecord(payload.spend_control) ? payload.spend_control : undefined;
203
+ return spendControl?.reached !== true;
204
+ }
205
+
164
206
  function parseUsagePayload(payload: unknown): ParsedUsage | null {
165
207
  if (!isRecord(payload)) return null;
166
208
  const planType = typeof payload.plan_type === "string" ? payload.plan_type : undefined;
@@ -170,13 +212,15 @@ function parseUsagePayload(payload: unknown): ParsedUsage | null {
170
212
  .map(parseAdditionalRateLimit)
171
213
  .filter((value): value is ParsedAdditionalUsage => value !== null);
172
214
  if (!rateLimit && additional.length === 0) return null;
215
+ const planLimitReached = rateLimit ? toBoolean(rateLimit.limit_reached) : undefined;
173
216
  const parsed: ParsedUsage = {
174
217
  planType,
175
218
  allowed: rateLimit ? toBoolean(rateLimit.allowed) : undefined,
176
- limitReached: rateLimit ? toBoolean(rateLimit.limit_reached) : undefined,
219
+ limitReached: planLimitReached,
177
220
  primary: rateLimit ? parseUsageWindow(rateLimit.primary_window) : undefined,
178
221
  secondary: rateLimit ? parseUsageWindow(rateLimit.secondary_window) : undefined,
179
222
  additional,
223
+ creditOverage: hasPlanCreditOverage(payload, planLimitReached),
180
224
  raw: payload as CodexUsagePayload,
181
225
  };
182
226
  if (
@@ -273,6 +317,17 @@ function buildUsageStatus(args: { usedFraction?: number; explicitlyAllowed: bool
273
317
  return "ok";
274
318
  }
275
319
 
320
+ /**
321
+ * Whether Codex will still serve this meter: an explicit positive verdict, or
322
+ * credits covering overage of a spent plan window. The credit override needs
323
+ * `limitReached === true`; a refusal for any other reason is not something a
324
+ * balance can pay for.
325
+ */
326
+ function isCodexRequestAllowed(args: { allowed?: boolean; limitReached?: boolean; creditOverage?: boolean }): boolean {
327
+ if (args.creditOverage === true && args.limitReached === true) return true;
328
+ return args.allowed === true && args.limitReached === false;
329
+ }
330
+
276
331
  function buildUsageLimit(args: {
277
332
  key: "primary" | "secondary";
278
333
  window: ParsedUsageWindow;
@@ -280,6 +335,7 @@ function buildUsageLimit(args: {
280
335
  planType?: string;
281
336
  allowed?: boolean;
282
337
  limitReached?: boolean;
338
+ creditOverage?: boolean;
283
339
  nowMs: number;
284
340
  }): UsageLimit {
285
341
  const usageWindow = buildUsageWindow(args.window, args.key, args.nowMs);
@@ -296,11 +352,12 @@ function buildUsageLimit(args: {
296
352
  amount,
297
353
  // The shared account-level rejection flag cannot identify which window
298
354
  // is binding, but an explicit positive verdict applies to both windows.
299
- // Preserve 100% as a warning when Codex still allows requests; live
300
- // usage_limit_reached responses remain authoritative for blocking.
355
+ // Preserve 100% as a warning when Codex still allows requests — either
356
+ // explicitly, or because credits fund overage past the plan window.
357
+ // Live usage_limit_reached responses remain authoritative for blocking.
301
358
  status: buildUsageStatus({
302
359
  usedFraction: amount.usedFraction,
303
- explicitlyAllowed: args.allowed === true && args.limitReached === false,
360
+ explicitlyAllowed: isCodexRequestAllowed(args),
304
361
  }),
305
362
  };
306
363
  }
@@ -354,9 +411,11 @@ function buildAdditionalUsageLimit(args: {
354
411
  amount,
355
412
  // A positive meter verdict is authoritative even when the advisory
356
413
  // percentage rounds to 100; negative shared verdicts remain window-local.
414
+ // Plan credits are deliberately not passed here: this meter is a separate
415
+ // allowance, and nothing in the payload says a balance funds its overage.
357
416
  status: buildUsageStatus({
358
417
  usedFraction: amount.usedFraction,
359
- explicitlyAllowed: args.allowed === true && args.limitReached === false,
418
+ explicitlyAllowed: isCodexRequestAllowed(args),
360
419
  }),
361
420
  };
362
421
  }
@@ -367,7 +426,11 @@ function buildAdditionalUsageLimit(args: {
367
426
  * ingesting them lets credential selection block an exhausted account before
368
427
  * the next request burns a wire 429.
369
428
  */
370
- export function parseCodexRateLimitHeaders(headers: Record<string, string>, now = Date.now()): UsageReport | null {
429
+ export function parseCodexRateLimitHeaders(
430
+ headers: Record<string, string>,
431
+ now = Date.now(),
432
+ context?: { responseStatus?: number },
433
+ ): UsageReport | null {
371
434
  const parseWindow = (key: "primary" | "secondary"): ParsedUsageWindow | undefined => {
372
435
  const usedPercent = toNumber(headers[`x-codex-${key}-used-percent`]);
373
436
  if (usedPercent === undefined) return undefined;
@@ -383,8 +446,11 @@ export function parseCodexRateLimitHeaders(headers: Record<string, string>, now
383
446
  const secondary = parseWindow("secondary");
384
447
  if (!primary && !secondary) return null;
385
448
  const limits: UsageLimit[] = [];
386
- if (primary) limits.push(buildUsageLimit({ key: "primary", window: primary, nowMs: now }));
387
- if (secondary) limits.push(buildUsageLimit({ key: "secondary", window: secondary, nowMs: now }));
449
+ const requestSucceeded =
450
+ context?.responseStatus !== undefined && context.responseStatus >= 200 && context.responseStatus < 300;
451
+ const verdict = requestSucceeded ? { allowed: true, limitReached: false } : {};
452
+ if (primary) limits.push(buildUsageLimit({ key: "primary", window: primary, ...verdict, nowMs: now }));
453
+ if (secondary) limits.push(buildUsageLimit({ key: "secondary", window: secondary, ...verdict, nowMs: now }));
388
454
  return {
389
455
  provider: "openai-codex",
390
456
  fetchedAt: now,
@@ -393,6 +459,21 @@ export function parseCodexRateLimitHeaders(headers: Record<string, string>, now
393
459
  };
394
460
  }
395
461
 
462
+ /**
463
+ * Plan meter verdict as credential selection should see it. Credits funding
464
+ * overage flip a plan-level rejection back to serving, which is what lets a
465
+ * stale usage-limit block self-heal instead of parking the account until reset.
466
+ * Only the plan meter takes this override — {@link hasPlanCreditOverage}.
467
+ */
468
+ function buildPlanMeterState(
469
+ allowed: boolean | undefined,
470
+ limitReached: boolean | undefined,
471
+ creditOverage: boolean,
472
+ ): { allowed?: boolean; limitReached?: boolean } {
473
+ if (!creditOverage) return { allowed, limitReached };
474
+ return { allowed: true, limitReached: false };
475
+ }
476
+
396
477
  export const openaiCodexUsageProvider: UsageProvider = {
397
478
  id: "openai-codex",
398
479
  supports(params: UsageFetchParams): boolean {
@@ -444,9 +525,10 @@ export const openaiCodexUsageProvider: UsageProvider = {
444
525
  parsed?.planType ??
445
526
  (isRecord(payload) && typeof payload.plan_type === "string" ? payload.plan_type : undefined);
446
527
 
528
+ const creditOverage = parsed?.creditOverage === true;
447
529
  const limits: UsageLimit[] = [];
448
530
  const meterStates: Record<string, { allowed?: boolean; limitReached?: boolean }> = {
449
- chat: { allowed: parsed?.allowed, limitReached: parsed?.limitReached },
531
+ chat: buildPlanMeterState(parsed?.allowed, parsed?.limitReached, creditOverage),
450
532
  };
451
533
  if (parsed?.primary) {
452
534
  limits.push(
@@ -457,6 +539,7 @@ export const openaiCodexUsageProvider: UsageProvider = {
457
539
  planType,
458
540
  allowed: parsed.allowed,
459
541
  limitReached: parsed.limitReached,
542
+ creditOverage,
460
543
  nowMs,
461
544
  }),
462
545
  );
@@ -470,6 +553,7 @@ export const openaiCodexUsageProvider: UsageProvider = {
470
553
  planType,
471
554
  allowed: parsed.allowed,
472
555
  limitReached: parsed.limitReached,
556
+ creditOverage,
473
557
  nowMs,
474
558
  }),
475
559
  );
@@ -548,8 +632,7 @@ export const openaiCodexUsageProvider: UsageProvider = {
548
632
  ...(resetCredits ? { resetCredits } : {}),
549
633
  metadata: {
550
634
  planType,
551
- allowed: parsed?.allowed,
552
- limitReached: parsed?.limitReached,
635
+ ...buildPlanMeterState(parsed?.allowed, parsed?.limitReached, creditOverage),
553
636
  email,
554
637
  accountId,
555
638
  meterStates,
package/src/usage.ts CHANGED
@@ -54,6 +54,8 @@ export interface UsageScope {
54
54
  tier?: string;
55
55
  windowId?: string;
56
56
  shared?: boolean;
57
+ /** Stable identity shared by routing-specific copies of one upstream quota. */
58
+ sharedGroup?: string;
57
59
  }
58
60
 
59
61
  /** Normalized limit entry for a single window or quota bucket. */
@@ -271,6 +273,7 @@ export const usageScopeSchema = type({
271
273
  "tier?": "string",
272
274
  "windowId?": "string",
273
275
  "shared?": "boolean",
276
+ "sharedGroup?": "string",
274
277
  });
275
278
 
276
279
  export const usageLimitSchema = type({
@@ -353,7 +356,11 @@ export interface UsageProvider {
353
356
  id: Provider;
354
357
  fetchUsage(params: UsageFetchParams, ctx: UsageFetchContext): Promise<UsageReport | null>;
355
358
  /** Parse provider rate-limit response headers (lowercased keys) into a usage report, if supported. */
356
- parseRateLimitHeaders?(headers: Record<string, string>, now?: number): UsageReport | null;
359
+ parseRateLimitHeaders?(
360
+ headers: Record<string, string>,
361
+ now?: number,
362
+ context?: { responseStatus?: number },
363
+ ): UsageReport | null;
357
364
  supports?(params: UsageFetchParams): boolean;
358
365
  /** True when fetchUsage contacts upstream and can authenticate the credential for health checks. */
359
366
  validatesCredentials?: boolean;
@@ -141,3 +141,60 @@ export type SyntheticUserCarrier = object & { [kSyntheticUser]?: boolean };
141
141
  export function isSyntheticUser(message: SyntheticUserCarrier | null | undefined): boolean {
142
142
  return message?.[kSyntheticUser] === true;
143
143
  }
144
+
145
+ /**
146
+ * Marks a message synthesized by a per-call context transform rather than
147
+ * loaded from persisted conversation history.
148
+ *
149
+ * Prompt-cache boundaries must skip these messages: their content is rebuilt
150
+ * for each request and cannot anchor a prefix reused by the next turn.
151
+ * Symbol-keyed so the marker never persists or reaches the provider wire.
152
+ */
153
+ export const kPerCallContextMessage = Symbol("agent.message.perCallContext");
154
+
155
+ /** Carries per-call context provenance without exposing a string-keyed property. */
156
+ export type PerCallContextMessageCarrier = object & { [kPerCallContextMessage]?: true };
157
+
158
+ /** Marks a message as synthesized for the current provider call. */
159
+ export function markPerCallContextMessage(message: PerCallContextMessageCarrier): void {
160
+ message[kPerCallContextMessage] = true;
161
+ }
162
+
163
+ /** Copies per-call context provenance to a converted or projected message. */
164
+ export function copyPerCallContextMessage(
165
+ target: PerCallContextMessageCarrier,
166
+ source: PerCallContextMessageCarrier,
167
+ ): void {
168
+ if (source[kPerCallContextMessage] === true) target[kPerCallContextMessage] = true;
169
+ }
170
+
171
+ /** True when a message was synthesized for the current provider call. */
172
+ export function isPerCallContextMessage(message: PerCallContextMessageCarrier | null | undefined): boolean {
173
+ return message?.[kPerCallContextMessage] === true;
174
+ }
175
+
176
+ /**
177
+ * Original history position carried by a context message clone.
178
+ *
179
+ * Object-spread transforms retain this symbol, allowing the extension runner
180
+ * to distinguish byte-identical historical copies from inserted messages.
181
+ */
182
+ export const kContextHistoryIndex = Symbol("agent.message.contextHistoryIndex");
183
+
184
+ /** Carries a context message's original history position. */
185
+ export type ContextHistoryIndexCarrier = object & { [kContextHistoryIndex]?: number };
186
+
187
+ /** Reads a context message's original history position. */
188
+ export function getContextHistoryIndex(message: ContextHistoryIndexCarrier | null | undefined): number | undefined {
189
+ return message?.[kContextHistoryIndex];
190
+ }
191
+
192
+ /** Records a context message's original history position. */
193
+ export function setContextHistoryIndex(message: ContextHistoryIndexCarrier, index: number): void {
194
+ message[kContextHistoryIndex] = index;
195
+ }
196
+
197
+ /** Removes context-history tracking before provider conversion. */
198
+ export function clearContextHistoryIndex(message: ContextHistoryIndexCarrier): void {
199
+ delete message[kContextHistoryIndex];
200
+ }
@@ -165,10 +165,30 @@ export function rewriteClinePassError(errorMessage: string, provider: string): s
165
165
  function sanitizeDump(dump: RawHttpRequestDump): RawHttpRequestDump {
166
166
  return {
167
167
  ...dump,
168
+ url: redactUrlQuery(dump.url),
168
169
  headers: redactHeaders(dump.headers),
169
170
  };
170
171
  }
171
172
 
173
+ /**
174
+ * Strips a persisted dump's query string entirely rather than picking sensitive
175
+ * params by name: a configurable `baseUrl` (e.g. Bedrock's gateway routing) can
176
+ * carry an arbitrary query-based credential the way `SENSITIVE_HEADER_PATTERN`
177
+ * matches arbitrary header names, and dumps exist to diagnose the request body,
178
+ * not the query.
179
+ */
180
+ function redactUrlQuery(url: string | undefined): string | undefined {
181
+ if (!url) return url;
182
+ try {
183
+ const parsed = new URL(url);
184
+ if (!parsed.search) return url;
185
+ parsed.search = "";
186
+ return `${parsed.toString()}[redacted-query]`;
187
+ } catch {
188
+ return url;
189
+ }
190
+ }
191
+
172
192
  function redactHeaders(headers: Record<string, string> | undefined): Record<string, string> | undefined {
173
193
  if (!headers) {
174
194
  return undefined;