@gajae-code/ai 0.12.8 → 0.12.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/CHANGELOG.md +19 -0
  2. package/dist/types/auth-storage.d.ts +25 -2
  3. package/dist/types/index.d.ts +1 -1
  4. package/dist/types/model-pricing.d.ts +3 -0
  5. package/dist/types/providers/composer-discipline.d.ts +29 -23
  6. package/dist/types/providers/openai-responses-shared.d.ts +1 -0
  7. package/dist/types/providers/register-builtins.d.ts +2 -2
  8. package/dist/types/types.d.ts +14 -6
  9. package/dist/types/utils/fallback-transport.d.ts +23 -0
  10. package/dist/types/utils/oauth/anthropic.d.ts +21 -2
  11. package/dist/types/utils/oauth/callback-server.d.ts +7 -0
  12. package/dist/types/utils/oauth/types.d.ts +11 -0
  13. package/package.json +2 -2
  14. package/src/auth-gateway/server.ts +6 -0
  15. package/src/auth-storage.ts +46 -5
  16. package/src/index.ts +1 -0
  17. package/src/model-manager.ts +4 -1
  18. package/src/model-pricing.ts +68 -0
  19. package/src/model-thinking.ts +23 -1
  20. package/src/models.json +122 -24
  21. package/src/models.ts +7 -4
  22. package/src/prompts/composer-bash-policy-recovery.md +1 -0
  23. package/src/prompts/cursor-composer-bash-policy-recovery.md +1 -0
  24. package/src/prompts/cursor-composer-edit-discipline.md +7 -0
  25. package/src/providers/composer-discipline.ts +54 -0
  26. package/src/providers/cursor.ts +2 -2
  27. package/src/providers/openai-completions.ts +23 -15
  28. package/src/providers/openai-responses-shared.ts +14 -3
  29. package/src/providers/openai-responses.ts +15 -8
  30. package/src/providers/register-builtins.ts +4 -4
  31. package/src/stream.ts +60 -5
  32. package/src/types.ts +16 -6
  33. package/src/utils/fallback-transport.ts +79 -6
  34. package/src/utils/idle-iterator.ts +2 -0
  35. package/src/utils/oauth/anthropic.ts +41 -8
  36. package/src/utils/oauth/callback-server.ts +64 -16
  37. package/src/utils/oauth/types.ts +12 -0
package/CHANGELOG.md CHANGED
@@ -2,10 +2,25 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.12.10] - 2026-08-03
6
+
7
+ ## [0.12.9] - 2026-08-03
8
+ ### Added
9
+
10
+ - Anthropic OAuth can now pair by pasting the authorization code Anthropic displays (`https://platform.claude.com/oauth/code/callback`) instead of waiting on `http://localhost:54545/callback`, so a browser with no network route back to the machine running gjc can complete the login. Opt in per login with `OAuthLoginOptions.manualCode`; the loopback flow stays the default and is unchanged. Callback flows can now opt out of binding a local listener entirely (`OAuthCallbackFlowOptions.skipCallbackServer`), which fails fast when no manual code handler is supplied instead of idling until the five-minute timeout. The hosted redirect is a hard-coded constant with no env or config override, so it cannot be repointed at an attacker-controlled collector.
11
+
12
+ ### Fixed
13
+
14
+ - Composer shell-policy failures now expose a stable structured marker plus provider-specific recovery guidance, while retaining recognition of prefix-only errors from older sessions. Cursor Composer requests use a native `read`/`grep`/`write`/`delete` discipline prompt rather than the generic hashline-tool vocabulary.
15
+ - Alibaba Token Plan streams now allow 600 seconds for the first semantic event, matching observed long-context TTFT above the previous 300-second cutoff. The outer lazy watchdog and both OpenAI transports share one provider fallback; OpenAI Completions also applies it before response headers, and Alibaba SDK connection timeouts from that pre-stream phase are normalized to the typed first-event failure so session retry policy does not replay the request as an unknown timeout.
16
+ - A plain `forbidden` failure no longer mutates credential state. `classifyFallbackTrigger` still returns the same `auth` class for HTTP 401 and 403, but now carries an `authDisposition` refinement of `"credential"` or `"forbidden"`. The refinement reads every code field (`openaiErrorCode`, `anthropicErrorType`, `providerCode`) and orders by specificity: a concrete credential fault wins, a `forbidden` in any field is otherwise terminal (so `{status: 401, providerCode: "forbidden"}` does not rotate), and the status decides only when no auth code is present. `transportFailureFacts` also reads `anthropicErrorType` back from its own key so re-normalizing already-built facts no longer drops it. `streamSimple` consults the disposition at both auth-capture exits — the error-event path and the thrown-error path, the latter unwrapping a nested `error.transportFailure` carrier that the shared `transportFailureFacts` extractor does not dereference — so a forbidden failure never reaches `onAuthError`, and `createAssistantAuthError` now preserves the structured transport facts on the callback error instead of reducing it to a status. The auth gateway's managed-failure bookkeeping likewise stops invalidating a credential on a forbidden response. Previously a single 403 could block an otherwise-healthy credential, and in a multi-credential pool could cycle through and block every row.
17
+ - `AuthStorage` gains `hasRuntimeCredentialSelector()` and `getSessionCredentialRowId()`. The first reports the `--credential` runtime pin, which lives in a different map from the `--api-key` override and previously had no accessor, so callers that must not rotate away from a pinned credential could not see it. The second returns the opaque stored row id for a session's current credential — never an email, account, project, or key material.
18
+
5
19
  ## [0.12.8] - 2026-08-02
6
20
  ### Added
7
21
 
8
22
  - Added read-only OpenCodex provider discovery with runtime-port resolution, identity-checked health probing, cached `/api/models` catalogs, raw wire model ids, and `/login opencodex` status reprobes without credential persistence.
23
+ - Added the Alibaba Token Plan `deepseek-v4-flash-0731` model with its 1M context, 384K output limit, OpenAI Completions routing, and documented low/high/max reasoning efforts.
9
24
 
10
25
  ### Changed
11
26
 
@@ -15,6 +30,10 @@
15
30
 
16
31
  - Closed the two remaining ingress holes behind bare `Request Blocked` failures on OpenAI codex models. (1) The chatgpt.com/backend-api pre-model gate rejects with an HTTP 400 bare-`detail` body (`{"detail": "Request blocked."}`) carrying no `error.*` envelope and no `code=invalid_prompt`, so `parseCodexError` surfaced an unexplained message, `isInvalidPromptError` and the codex non-retryable classification missed it, and the session-level `invalid_prompt` circuit breaker never attempted a repaired resend. `parseCodexError` now reads top-level `detail` (string or `{message}`) bodies and classifies a leading `Request blocked` message without an explicit provider code as `invalid_prompt`, surfacing `Request blocked (code=invalid_prompt)` so every existing invalid_prompt contract engages. (2) Outgoing tool definitions (descriptions and JSON-schema strings) bypassed every request-boundary sanitizer on both the OpenAI Responses and OpenAI-codex-responses transports, so a `<|channel|>`-quoting MCP/skill tool description poisoned every request on the session in a way no history repair could fix. Both `convertTools` paths now neutralize reserved control tokens across the whole tool payload via the shared idempotent zero-width-space insertion (ref openai/codex#35838).
17
32
 
33
+ ### Fixed
34
+
35
+ - Updated GPT-5.6 Sol, Terra, and Luna to current OpenAI Standard pricing, including Responses API cache-write attribution and full-request long-context pricing above 272K input tokens.
36
+
18
37
  ## [0.12.7] - 2026-07-31
19
38
 
20
39
  ## [0.12.6] - 2026-07-31
@@ -10,7 +10,7 @@
10
10
  import { Database } from "bun:sqlite";
11
11
  import type { Provider } from "./types";
12
12
  import type { CredentialRankingStrategy, UsageLogger, UsageProvider, UsageReport } from "./usage";
13
- import type { OAuthController, OAuthCredentials, OAuthProviderId } from "./utils/oauth/types";
13
+ import type { OAuthController, OAuthCredentials, OAuthLoginOptions, OAuthProviderId } from "./utils/oauth/types";
14
14
  export type ApiKeyCredential = {
15
15
  type: "api_key";
16
16
  key: string;
@@ -434,6 +434,29 @@ export declare class AuthStorage {
434
434
  removeRuntimeApiKey(provider: string): void;
435
435
  /** Whether a provider is currently authenticated by a runtime API-key override. */
436
436
  hasRuntimeApiKey(provider: string): boolean;
437
+ /**
438
+ * Whether credential selection for a provider is pinned to one stored row by
439
+ * a runtime selector (`--credential`).
440
+ *
441
+ * Distinct from {@link AuthStorage.hasRuntimeApiKey}: that reports the
442
+ * `--api-key` override, which lives in a different map and is mutually
443
+ * exclusive with a selector. Callers that must not rotate away from a pinned
444
+ * credential have to consult BOTH.
445
+ */
446
+ hasRuntimeCredentialSelector(provider: string): boolean;
447
+ /**
448
+ * Opaque stored row id of the credential this session is currently using.
449
+ *
450
+ * Deliberately non-identifying: the persisted primary key, never an email,
451
+ * account id, project id, or key material. Callers that need to correlate a
452
+ * credential across a session boundary use this instead of projecting
453
+ * personal metadata.
454
+ *
455
+ * Returns `undefined` when the session has not been routed to a stored
456
+ * credential yet, or when it authenticated through an env key or fallback
457
+ * resolver rather than a stored row.
458
+ */
459
+ getSessionCredentialRowId(provider: string, sessionId?: string): number | undefined;
437
460
  /**
438
461
  * Register a per-provider API key sourced from user configuration
439
462
  * (e.g. `models.yml` `providers.<name>.apiKey`). Higher priority than
@@ -525,7 +548,7 @@ export declare class AuthStorage {
525
548
  message: string;
526
549
  placeholder?: string;
527
550
  }) => Promise<string>;
528
- }): Promise<void>;
551
+ }, options?: OAuthLoginOptions): Promise<void>;
529
552
  /**
530
553
  * Logout from a provider.
531
554
  */
@@ -46,7 +46,7 @@ export * from "./utils/event-stream";
46
46
  export * from "./utils/fallback-transport";
47
47
  export * from "./utils/h2-fetch";
48
48
  export * from "./utils/oauth";
49
- export type { OAuthCredentials, OAuthProvider, OAuthProviderId, OAuthProviderInfo, } from "./utils/oauth/types";
49
+ export type { OAuthCredentials, OAuthLoginOptions, OAuthProvider, OAuthProviderId, OAuthProviderInfo, } from "./utils/oauth/types";
50
50
  export * from "./utils/overflow";
51
51
  export * from "./utils/retry";
52
52
  export * from "./utils/schema";
@@ -0,0 +1,3 @@
1
+ import type { Api, Model, ModelCost } from "./types";
2
+ export declare function getOpenAIModelCost<TApi extends Api>(model: Model<TApi>, inputTokens: number): ModelCost | undefined;
3
+ export declare function applyOpenAIModelPricing<TApi extends Api>(model: Model<TApi>): void;
@@ -1,26 +1,32 @@
1
+ export declare function isComposerHarnessModel(modelId: string): boolean;
2
+ /** Stable text contract for a local shell rejection caused by Composer file-I/O discipline. */
3
+ export declare const COMPOSER_BASH_POLICY_ERROR_PREFIX = "Composer bash policy blocked repository file I/O.";
4
+ export declare const COMPOSER_BASH_POLICY_ERROR_CODE = "composer-bash-policy:repository-file-io";
5
+ export type ComposerBashPolicyToolSurface = "generic" | "cursor";
1
6
  /**
2
- * Anchor/edit discipline for composer-harness models (xai grok-composer-*,
3
- * cursor composer-*).
4
- *
5
- * Composer models are trained on a proprietary coding-agent harness
6
- * (Cursor / Grok Build) and carry habits that break this agent's hashline
7
- * edit workflow when driven through a generic provider. Observed in live
8
- * sessions with grok-composer-2.5-fast:
9
- *
10
- * - they print files with shell commands (`sed -n`, `cat`, `grep -n`) or
11
- * python heredocs whose output carries NO line anchors, then FABRICATE the
12
- * 2-char anchor hash the edit tool requires (e.g. guessed "617hp" where
13
- * the file had "617ca" → "Edit rejected: N anchors do not match");
14
- * - they mutate files out-of-band via python heredocs (pathlib write_text /
15
- * str.replace), which invalidates every previously seen anchor and defeats
16
- * the read-cache snapshot that powers stale-anchor recovery;
17
- * - they arithmetically renumber anchors after their own edits instead of
18
- * copying them from the latest tool output;
19
- * - they leak reasoning prose into heredoc bodies, producing shell/python
20
- * syntax errors.
21
- *
22
- * This prompt is the per-request countermeasure, pinned ahead of the host
23
- * system prompt on openai-completions, openai-responses, and cursor RPC paths.
7
+ * Format the model-visible policy rejection with a stable marker and the tool
8
+ * vocabulary the model actually receives on this provider surface.
24
9
  */
25
- export declare function isComposerHarnessModel(modelId: string): boolean;
10
+ export declare function formatComposerBashPolicyError(surface?: ComposerBashPolicyToolSurface): string;
11
+ /**
12
+ * Matches both the structured current error and the original prefix so a
13
+ * resumed session can recover after an upgrade without string-version skew.
14
+ */
15
+ export declare function isComposerBashPolicyBlockedError(text: string): boolean;
16
+ /**
17
+ * Matches only errors emitted directly by the current policy implementation.
18
+ * Live recovery must use this strict form so failed shell output that merely
19
+ * quotes a policy error cannot masquerade as the policy gate itself.
20
+ */
21
+ export declare function isCurrentComposerBashPolicyBlockedError(text: string): boolean;
22
+ /** One bounded, tool-enabled retry instruction for generic Composer agent loops. */
23
+ export declare const COMPOSER_BASH_POLICY_RECOVERY_PROMPT: string;
24
+ /** One bounded, tool-enabled retry instruction for Cursor's native remote tool surface. */
25
+ export declare const CURSOR_COMPOSER_BASH_POLICY_RECOVERY_PROMPT: string;
26
26
  export declare const COMPOSER_EDIT_DISCIPLINE_PROMPT = "File-editing discipline for this Composer harness (this OVERRIDES contrary habits from your training):\n\n- Discover file names ONLY with the find tool; search file contents ONLY with the search tool; read file bodies or line ranges ONLY with the read tool. NEVER inspect repository files through shell commands (ls, find, fd, cat, sed, awk, grep, rg, head, tail, less, more) or scripts \u2014 that output carries no hashline anchors and bypasses the agent's safety limits.\n- Modify files ONLY with the edit/write tools. NEVER mutate files through shell redirection, tee, sed -i, perl -pi, inline python/node/bun scripts, or other out-of-band writes \u2014 those writes invalidate every known anchor and break edit recovery.\n- A line anchor (e.g. \"42sr\") is a line number plus a 2-char content hash. You CANNOT compute the hash yourself: copy anchors verbatim from the MOST RECENT read/search/edit output of that exact file. NEVER guess, renumber, or arithmetically shift an anchor.\n- After ANY edit to a file (including your own), anchors you saw earlier are stale. Re-read the edited region, or copy the fresh anchors printed in the edit result, before issuing the next edit.\n- If an edit is rejected with \"anchors do not match\", the rejection message prints the current lines WITH fresh anchors. Retry using exactly those printed anchors.\n- Tool-call arguments must be the exact JSON/schema object requested by the tool. Do not include Markdown, commentary, analysis text, or invented fields inside tool arguments.\n- Use bash only for terminal operations such as tests, builds, package scripts, and git commands. A shell command string must contain only the command itself; NEVER interleave reasoning or commentary into command strings or heredocs.";
27
+ /**
28
+ * Cursor executes a different native tool vocabulary from the generic agent
29
+ * loop. Keep this prompt separate so Composer is never told to call `edit`,
30
+ * `find`, or `search` when those names are unavailable remotely.
31
+ */
32
+ export declare const CURSOR_COMPOSER_EDIT_DISCIPLINE_PROMPT: string;
@@ -96,6 +96,7 @@ export declare function populateResponsesUsageFromResponse(output: AssistantMess
96
96
  total_tokens?: number | null;
97
97
  input_tokens_details?: {
98
98
  cached_tokens?: number | null;
99
+ cache_write_tokens?: number | null;
99
100
  } | null;
100
101
  output_tokens_details?: {
101
102
  reasoning_tokens?: number | null;
@@ -20,8 +20,8 @@ export declare function setBedrockProviderModule(module: BedrockProviderModule):
20
20
  /**
21
21
  * Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
22
22
  * A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
23
- * otherwise providers known to have slow first events get a five-minute floor
24
- * matching their inner provider-level override. Returns `undefined` for
23
+ * otherwise providers known to have slow first events use the same centralized
24
+ * fallback as their inner provider-level watchdog. Returns `undefined` for
25
25
  * providers that should use the shared default.
26
26
  */
27
27
  export declare function resolveLazyStreamFirstEventFallbackMs(provider: string, configuredFallbackMs?: number): number | undefined;
@@ -827,6 +827,17 @@ export interface ModelRequestTransform {
827
827
  /** Extra request body fields merged after provider defaults; protected core request keys are ignored. */
828
828
  extraBody?: Record<string, unknown>;
829
829
  }
830
+ export interface ModelCost {
831
+ input: number;
832
+ output: number;
833
+ cacheRead: number;
834
+ cacheWrite: number;
835
+ }
836
+ export interface LongContextPricing {
837
+ /** Input-token count above which the long-context rates apply to the full request. */
838
+ threshold: number;
839
+ cost: ModelCost;
840
+ }
830
841
  export interface Model<TApi extends Api = any> {
831
842
  id: string;
832
843
  name: string;
@@ -843,12 +854,9 @@ export interface Model<TApi extends Api = any> {
843
854
  * provider/id heuristics.
844
855
  */
845
856
  output?: ("text" | "image")[];
846
- cost: {
847
- input: number;
848
- output: number;
849
- cacheRead: number;
850
- cacheWrite: number;
851
- };
857
+ cost: ModelCost;
858
+ /** Optional long-context rates selected from the request's total input-token count. */
859
+ longContextPricing?: LongContextPricing;
852
860
  /** Premium Copilot requests charged per user-initiated request (defaults to 1). */
853
861
  premiumMultiplier?: number;
854
862
  contextWindow: number;
@@ -1,7 +1,23 @@
1
1
  export type FallbackTriggerClass = "rate_limit" | "quota" | "auth" | "server" | "unknown" | "other";
2
+ /**
3
+ * Refinement of an `auth` trigger.
4
+ *
5
+ * The transport deliberately collapses HTTP 401 and 403 into a single `auth`
6
+ * class, but the two demand opposite handling: a credential problem may be
7
+ * recoverable by trying a different stored credential, whereas a plain
8
+ * `forbidden` is an authorization or configuration defect that rotation would
9
+ * only hide — it would cycle and block every otherwise-healthy credential.
10
+ *
11
+ * This is a refinement rather than a new {@link FallbackTriggerClass} member so
12
+ * every existing `trigger.class === "auth"` consumer keeps compiling and keeps
13
+ * its current behavior until it explicitly opts into the distinction.
14
+ */
15
+ export type AuthDisposition = "credential" | "forbidden";
2
16
  export interface FallbackTrigger {
3
17
  class: FallbackTriggerClass;
4
18
  retryAfterMs?: number;
19
+ /** Present only when `class === "auth"`. */
20
+ authDisposition?: AuthDisposition;
5
21
  }
6
22
  /** Stable code for streams that time out before producing semantic progress. */
7
23
  export declare const STREAM_FIRST_EVENT_TIMEOUT_PROVIDER_CODE = "stream_first_event_timeout";
@@ -66,3 +82,10 @@ export declare function transportFailureFacts(error: unknown, capturedResponse?:
66
82
  }): TransportFailureFacts | undefined;
67
83
  /** Classifies only typed upstream transport facts without consuming response bodies. */
68
84
  export declare function classifyFallbackTrigger(errorOrFacts: TransportFailureFacts | FallbackTriggerInput | unknown): FallbackTrigger;
85
+ /**
86
+ * True when a failure is an `auth` failure that must NOT rotate credentials.
87
+ *
88
+ * Callers that mutate credential state on auth failures should consult this
89
+ * first so a plain `forbidden` cannot block otherwise-healthy credentials.
90
+ */
91
+ export declare function isForbiddenAuthFailure(errorOrFacts: TransportFailureFacts | FallbackTriggerInput | unknown): boolean;
@@ -3,9 +3,28 @@
3
3
  */
4
4
  import { OAuthCallbackFlow } from "./callback-server";
5
5
  import type { OAuthController, OAuthCredentials } from "./types";
6
+ /**
7
+ * Redirect target for the paste-a-code login. Anthropic renders the
8
+ * authorization code on this page instead of redirecting into this machine, so
9
+ * a gjc running over SSH, in a container, or on a headless box can be paired
10
+ * from a browser that has no route back to `localhost:54545`.
11
+ *
12
+ * Deliberately a hard-coded constant rather than an env/config override: this
13
+ * is where the authorization code is delivered, so making it injectable would
14
+ * turn any writable environment into an auth-code exfiltration channel.
15
+ */
16
+ export declare const ANTHROPIC_MANUAL_REDIRECT_URI = "https://platform.claude.com/oauth/code/callback";
17
+ export interface AnthropicOAuthFlowOptions {
18
+ /**
19
+ * Pair by pasting the code Anthropic displays instead of waiting on a local
20
+ * `localhost:54545` callback. Use when the browser completing the login has
21
+ * no network route back to the machine running gjc.
22
+ */
23
+ manualCode?: boolean;
24
+ }
6
25
  export declare class AnthropicOAuthFlow extends OAuthCallbackFlow {
7
26
  #private;
8
- constructor(ctrl: OAuthController);
27
+ constructor(ctrl: OAuthController, options?: AnthropicOAuthFlowOptions);
9
28
  generateAuthUrl(state: string, redirectUri: string): Promise<{
10
29
  url: string;
11
30
  instructions?: string;
@@ -15,7 +34,7 @@ export declare class AnthropicOAuthFlow extends OAuthCallbackFlow {
15
34
  /**
16
35
  * Login with Anthropic OAuth
17
36
  */
18
- export declare function loginAnthropic(ctrl: OAuthController): Promise<OAuthCredentials>;
37
+ export declare function loginAnthropic(ctrl: OAuthController, options?: AnthropicOAuthFlowOptions): Promise<OAuthCredentials>;
19
38
  /**
20
39
  * Refresh Anthropic OAuth token
21
40
  */
@@ -11,6 +11,13 @@ export interface OAuthCallbackFlowOptions {
11
11
  callbackBindHostname?: string;
12
12
  /** Exact redirect URI advertised to the provider; disables port fallback. */
13
13
  redirectUri?: string;
14
+ /**
15
+ * Do not bind a local listener at all. The provider redirects somewhere this
16
+ * process cannot observe (a hosted "copy this code" page, a custom protocol),
17
+ * so the code arrives by paste instead. Requires both `redirectUri` and an
18
+ * `onManualCodeInput` handler on the controller.
19
+ */
20
+ skipCallbackServer?: boolean;
14
21
  }
15
22
  /**
16
23
  * Abstract base class for OAuth flows with local callback servers.
@@ -23,6 +23,17 @@ export interface OAuthProviderInfo {
23
23
  name: string;
24
24
  available: boolean;
25
25
  }
26
+ /** Per-login switches that change how the authorization code is delivered. */
27
+ export interface OAuthLoginOptions {
28
+ /**
29
+ * Pair by pasting the authorization code the provider displays instead of
30
+ * waiting on a local loopback callback. Set when the browser completing the
31
+ * login has no network route back to the machine running gjc (SSH, remote
32
+ * container, headless host). Providers without a paste-a-code redirect
33
+ * ignore it.
34
+ */
35
+ manualCode?: boolean;
36
+ }
26
37
  export interface OAuthController {
27
38
  onAuth?(info: OAuthAuthInfo): void;
28
39
  onProgress?(message: string): void;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.12.8",
4
+ "version": "0.12.10",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -40,7 +40,7 @@
40
40
  "dependencies": {
41
41
  "@anthropic-ai/sdk": "^0.94.0",
42
42
  "@bufbuild/protobuf": "^2.12.0",
43
- "@gajae-code/utils": "0.12.8",
43
+ "@gajae-code/utils": "0.12.10",
44
44
  "openai": "^6.36.0",
45
45
  "partial-json": "^0.1.7",
46
46
  "zod": "4.4.3"
@@ -279,6 +279,12 @@ async function markManagedGatewayCredentialFailure(
279
279
  ): Promise<void> {
280
280
  const trigger = classifyFallbackTrigger(error);
281
281
  try {
282
+ if (trigger.class === "auth" && trigger.authDisposition === "forbidden") {
283
+ // A plain `forbidden` is an authorization or configuration defect.
284
+ // Blocking the credential here would hide it and would cycle through
285
+ // every otherwise-healthy row in a multi-credential pool.
286
+ return;
287
+ }
282
288
  if (trigger.class === "auth") {
283
289
  await storage.invalidateCredentialMatching(model.provider, apiKey, signal);
284
290
  } else if (trigger.class === "quota" || trigger.class === "rate_limit") {
@@ -37,7 +37,13 @@ import { getOAuthApiKey, getOAuthProvider, refreshOAuthToken, resolveOAuthStorag
37
37
  import { loginDeepInfra } from "./utils/oauth/deepinfra";
38
38
  import { loginDeepSeek } from "./utils/oauth/deepseek";
39
39
  import { loginOpenAICodexDevice } from "./utils/oauth/openai-codex";
40
- import type { OAuthController, OAuthCredentials, OAuthProvider, OAuthProviderId } from "./utils/oauth/types";
40
+ import type {
41
+ OAuthController,
42
+ OAuthCredentials,
43
+ OAuthLoginOptions,
44
+ OAuthProvider,
45
+ OAuthProviderId,
46
+ } from "./utils/oauth/types";
41
47
 
42
48
  // ─────────────────────────────────────────────────────────────────────────────
43
49
  // Credential Types
@@ -1089,6 +1095,37 @@ export class AuthStorage {
1089
1095
  return Boolean(this.#runtimeOverrides.get(provider));
1090
1096
  }
1091
1097
 
1098
+ /**
1099
+ * Whether credential selection for a provider is pinned to one stored row by
1100
+ * a runtime selector (`--credential`).
1101
+ *
1102
+ * Distinct from {@link AuthStorage.hasRuntimeApiKey}: that reports the
1103
+ * `--api-key` override, which lives in a different map and is mutually
1104
+ * exclusive with a selector. Callers that must not rotate away from a pinned
1105
+ * credential have to consult BOTH.
1106
+ */
1107
+ hasRuntimeCredentialSelector(provider: string): boolean {
1108
+ return this.#runtimeCredentialSelectors.has(resolveOAuthStorageProvider(provider));
1109
+ }
1110
+
1111
+ /**
1112
+ * Opaque stored row id of the credential this session is currently using.
1113
+ *
1114
+ * Deliberately non-identifying: the persisted primary key, never an email,
1115
+ * account id, project id, or key material. Callers that need to correlate a
1116
+ * credential across a session boundary use this instead of projecting
1117
+ * personal metadata.
1118
+ *
1119
+ * Returns `undefined` when the session has not been routed to a stored
1120
+ * credential yet, or when it authenticated through an env key or fallback
1121
+ * resolver rather than a stored row.
1122
+ */
1123
+ getSessionCredentialRowId(provider: string, sessionId?: string): number | undefined {
1124
+ const session = this.#getSessionCredential(provider, sessionId);
1125
+ if (!session) return undefined;
1126
+ return this.#getStoredCredentials(provider)[session.index]?.id;
1127
+ }
1128
+
1092
1129
  /**
1093
1130
  * Register a per-provider API key sourced from user configuration
1094
1131
  * (e.g. `models.yml` `providers.<name>.apiKey`). Higher priority than
@@ -1880,6 +1917,7 @@ export class AuthStorage {
1880
1917
  /** onPrompt is required for some providers (github-copilot, OpenAI code provider) */
1881
1918
  onPrompt: (prompt: { message: string; placeholder?: string }) => Promise<string>;
1882
1919
  },
1920
+ options: OAuthLoginOptions = {},
1883
1921
  ): Promise<void> {
1884
1922
  let credentials: OAuthCredentials;
1885
1923
  const saveApiKeyCredential = async (apiKey: string): Promise<void> => {
@@ -1895,10 +1933,13 @@ export class AuthStorage {
1895
1933
  }
1896
1934
  case "anthropic": {
1897
1935
  const { loginAnthropic } = await import("./utils/oauth/anthropic");
1898
- credentials = await loginAnthropic({
1899
- ...ctrl,
1900
- onManualCodeInput: ctrl.onManualCodeInput ?? manualCodeInput,
1901
- });
1936
+ credentials = await loginAnthropic(
1937
+ {
1938
+ ...ctrl,
1939
+ onManualCodeInput: ctrl.onManualCodeInput ?? manualCodeInput,
1940
+ },
1941
+ { manualCode: options.manualCode },
1942
+ );
1902
1943
  break;
1903
1944
  }
1904
1945
  case "alibaba-token-plan": {
package/src/index.ts CHANGED
@@ -48,6 +48,7 @@ export * from "./utils/h2-fetch";
48
48
  export * from "./utils/oauth";
49
49
  export type {
50
50
  OAuthCredentials,
51
+ OAuthLoginOptions,
51
52
  OAuthProvider,
52
53
  OAuthProviderId,
53
54
  OAuthProviderInfo,
@@ -406,9 +406,12 @@ function normalizeModelList<TApi extends Api>(value: unknown): Model<TApi>[] {
406
406
  const models: Model<TApi>[] = [];
407
407
  for (const item of value) {
408
408
  if (isModelLike(item) && !isRetiredModel(item)) {
409
- models.push(enrichModelThinking(item as Model<TApi>));
409
+ const model = enrichModelThinking(item as Model<TApi>);
410
+ model.longContextPricing = undefined;
411
+ models.push(model);
410
412
  }
411
413
  }
414
+ applyGeneratedModelPolicies(models as Model<Api>[]);
412
415
  return applyFinalCodexGpt56ContextCap(models);
413
416
  }
414
417
 
@@ -0,0 +1,68 @@
1
+ import type { Api, LongContextPricing, Model, ModelCost } from "./types";
2
+
3
+ interface TieredPricing {
4
+ cost: ModelCost;
5
+ longContextPricing: LongContextPricing;
6
+ }
7
+
8
+ const LONG_CONTEXT_THRESHOLD = 272_000;
9
+
10
+ const GPT_5_6_SOL_PRICING: TieredPricing = {
11
+ cost: { input: 5, output: 30, cacheRead: 0.5, cacheWrite: 6.25 },
12
+ longContextPricing: {
13
+ threshold: LONG_CONTEXT_THRESHOLD,
14
+ cost: { input: 10, output: 45, cacheRead: 1, cacheWrite: 12.5 },
15
+ },
16
+ };
17
+
18
+ // OpenAI Standard pricing: https://developers.openai.com/api/docs/pricing
19
+ const OPENAI_GPT_5_6_PRICING: ReadonlyMap<string, TieredPricing> = new Map([
20
+ ["gpt-5.6", GPT_5_6_SOL_PRICING],
21
+ ["gpt-5.6-sol", GPT_5_6_SOL_PRICING],
22
+ [
23
+ "gpt-5.6-terra",
24
+ {
25
+ cost: { input: 2, output: 12, cacheRead: 0.2, cacheWrite: 2.5 },
26
+ longContextPricing: {
27
+ threshold: LONG_CONTEXT_THRESHOLD,
28
+ cost: { input: 4, output: 18, cacheRead: 0.4, cacheWrite: 5 },
29
+ },
30
+ },
31
+ ],
32
+ [
33
+ "gpt-5.6-luna",
34
+ {
35
+ cost: { input: 0.2, output: 1.2, cacheRead: 0.02, cacheWrite: 0.25 },
36
+ longContextPricing: {
37
+ threshold: LONG_CONTEXT_THRESHOLD,
38
+ cost: { input: 0.4, output: 1.8, cacheRead: 0.04, cacheWrite: 0.5 },
39
+ },
40
+ },
41
+ ],
42
+ ]);
43
+
44
+ export function getOpenAIModelCost<TApi extends Api>(model: Model<TApi>, inputTokens: number): ModelCost | undefined {
45
+ if (model.provider !== "openai" && model.provider !== "openai-codex") {
46
+ return undefined;
47
+ }
48
+ const pricing = OPENAI_GPT_5_6_PRICING.get(model.id);
49
+ if (!pricing) {
50
+ return undefined;
51
+ }
52
+ return inputTokens > pricing.longContextPricing.threshold ? pricing.longContextPricing.cost : pricing.cost;
53
+ }
54
+
55
+ export function applyOpenAIModelPricing<TApi extends Api>(model: Model<TApi>): void {
56
+ if (model.provider !== "openai" && model.provider !== "openai-codex") {
57
+ return;
58
+ }
59
+ const pricing = OPENAI_GPT_5_6_PRICING.get(model.id);
60
+ if (!pricing) {
61
+ return;
62
+ }
63
+ model.cost = { ...pricing.cost };
64
+ model.longContextPricing = {
65
+ threshold: pricing.longContextPricing.threshold,
66
+ cost: { ...pricing.longContextPricing.cost },
67
+ };
68
+ }
@@ -1,4 +1,5 @@
1
1
  import { CODEX_GPT_5_6_CONTEXT_CAP, isCodexGpt56Tier, isCodexProductTransport } from "./context-cap-policy";
2
+ import { applyOpenAIModelPricing } from "./model-pricing";
2
3
  import { resolveOpenAICompat } from "./providers/openai-completions-compat";
3
4
  import type { Api, Model as ApiModel, ThinkingConfig } from "./types";
4
5
  import { isClaudeForcedToolChoiceIncapableModelId } from "./utils/tool-choice-capability";
@@ -51,6 +52,7 @@ const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effo
51
52
  const GPT_5_6_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max];
52
53
  const GPT_5_5_DEFAULT_EFFORT = Effort.XHigh;
53
54
  const KIMI_K3_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High, Effort.Max];
55
+ const DEEPSEEK_V4_FLASH_0731_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High, Effort.Max];
54
56
 
55
57
  const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High];
56
58
  const CLOUDFLARE_AI_GATEWAY_BASE_URL = "https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic";
@@ -198,7 +200,12 @@ export function refreshModelThinking<TApi extends Api>(model: ApiModel<TApi>): A
198
200
  */
199
201
  export function applyGeneratedModelPolicies(models: ApiModel<Api>[]): void {
200
202
  for (let index = 0; index < models.length; index++) {
201
- const model = refreshModelThinking(models[index]!);
203
+ const source = models[index]!;
204
+ if (source.provider === "alibaba-token-plan" && source.id === "deepseek-v4-flash-0731") {
205
+ source.reasoning = true;
206
+ source.name = "DeepSeek V4 Flash 0731";
207
+ }
208
+ const model = refreshModelThinking(source);
202
209
  applyGeneratedModelPolicy(model);
203
210
  models[index] = model;
204
211
  }
@@ -377,6 +384,7 @@ function anthropicModelHasRealXHighEffort<TApi extends Api>(model: ApiModel<TApi
377
384
  }
378
385
 
379
386
  function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
387
+ applyOpenAIModelPricing(model);
380
388
  const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
381
389
  if (copilotLimits) {
382
390
  model.contextWindow = copilotLimits.contextWindow;
@@ -437,6 +445,17 @@ function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
437
445
  if (model.provider === "zai" && model.id === "glm-5.2") {
438
446
  model.contextWindow = 1_000_000;
439
447
  }
448
+ if (model.provider === "alibaba-token-plan" && model.id === "deepseek-v4-flash-0731") {
449
+ model.contextWindow = 1_000_000;
450
+ model.maxTokens = 384_000;
451
+ model.compat = {
452
+ ...(model.compat ?? {}),
453
+ supportsDeveloperRole: false,
454
+ supportsReasoningEffort: true,
455
+ reasoningContentField: "reasoning_content",
456
+ requiresReasoningContentForToolCalls: true,
457
+ };
458
+ }
440
459
  // MiniMax-M3: MiniMax exposes a 1M context tier, but usage beyond 512K is
441
460
  // billed separately. Keep bundled/default metadata at the billing-safe 512K
442
461
  // unless an explicit paid-tier contract is added.
@@ -617,6 +636,9 @@ function inferSupportedEfforts<TApi extends Api>(parsedModel: ParsedModel, model
617
636
  if (model.provider === "kimi-code" && model.id === "k3") {
618
637
  return KIMI_K3_EFFORTS;
619
638
  }
639
+ if (model.provider === "alibaba-token-plan" && model.id === "deepseek-v4-flash-0731") {
640
+ return DEEPSEEK_V4_FLASH_0731_EFFORTS;
641
+ }
620
642
  switch (parsedModel.family) {
621
643
  case "openai":
622
644
  return inferOpenAISupportedEfforts(parsedModel);