@gajae-code/ai 0.10.2 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/CHANGELOG.md +13 -2
  2. package/dist/types/auth-gateway/server.d.ts +19 -0
  3. package/dist/types/auth-storage.d.ts +2 -0
  4. package/dist/types/context-cap-policy.d.ts +10 -0
  5. package/dist/types/index.d.ts +2 -0
  6. package/dist/types/providers/openai-codex/response-handler.d.ts +1 -0
  7. package/dist/types/providers/pi-native-server.d.ts +3 -3
  8. package/dist/types/types.d.ts +12 -4
  9. package/dist/types/utils/event-stream.d.ts +9 -3
  10. package/dist/types/utils/fallback-transport.d.ts +55 -0
  11. package/dist/types/utils/retry.d.ts +1 -0
  12. package/dist/types/utils.d.ts +17 -0
  13. package/package.json +2 -2
  14. package/src/auth-gateway/server.ts +164 -22
  15. package/src/auth-storage.ts +62 -40
  16. package/src/context-cap-policy.ts +59 -0
  17. package/src/index.ts +2 -0
  18. package/src/model-manager.ts +12 -7
  19. package/src/model-thinking.ts +7 -8
  20. package/src/providers/amazon-bedrock.ts +9 -1
  21. package/src/providers/anthropic.ts +6 -0
  22. package/src/providers/azure-openai-responses.ts +6 -1
  23. package/src/providers/google-gemini-cli.ts +26 -13
  24. package/src/providers/google-shared.ts +7 -1
  25. package/src/providers/ollama.ts +7 -1
  26. package/src/providers/openai-codex/response-handler.ts +11 -3
  27. package/src/providers/openai-codex-responses.ts +17 -3
  28. package/src/providers/openai-completions.ts +13 -2
  29. package/src/providers/openai-responses.ts +7 -2
  30. package/src/providers/pi-native-client.ts +24 -12
  31. package/src/providers/pi-native-server.ts +4 -3
  32. package/src/stream.ts +31 -2
  33. package/src/types.ts +27 -4
  34. package/src/utils/discovery/codex.ts +3 -12
  35. package/src/utils/event-stream.ts +138 -54
  36. package/src/utils/fallback-transport.ts +185 -0
  37. package/src/utils/retry.ts +2 -2
  38. package/src/utils.ts +19 -1
package/CHANGELOG.md CHANGED
@@ -2,6 +2,17 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.11.0] - 2026-07-15
6
+ ### Added
7
+
8
+ - Exported the canonical thinking-control mode runtime vocabulary so packed SDK consumers can validate provider metadata against the same public `@gajae-code/ai` contract.
9
+
10
+ ### Fixed
11
+
12
+ - Fixed frequent `Request blocked (code=invalid_prompt)` failures on gpt-5.6 (Sol/Terra/Luna) subagent, default-agent, and compaction turns (ref openai/codex#32028, oh-my-pi#5184). Leaked Harmony control-token markers (e.g. `<|channel|>analysis`) were only neutralized on the replayed-history payload path, so markers in assistant reasoning summaries, live-converted message/tool-output text, and user-authored content reached the OpenAI Responses and OpenAI-codex-responses transports verbatim and wedged the session (the poisoned item was re-sent every turn). Both transports now neutralize reserved control tokens across the entire outgoing `input` array at the request boundary via an idempotent zero-width-space insertion that keeps the text human-readable.
13
+
14
+ - Closed the remaining `Request blocked (code=invalid_prompt)` wedge on gpt-5.6 caused by header-form leaked Harmony markers. The reserved-control-token sanitizer only matched the simple `<|ident|>` shape, so a header-form marker carrying a recipient (e.g. `<|assistant to=functions.bash|>`) survived every sanitizer path (replay, request boundary, compaction) and kept re-poisoning history even after the earlier fixes. The pattern now also matches the scoped header grammar — a known Harmony role (`system`/`developer`/`user`/`assistant`/`tool`) plus a `to=<recipient>` assignment with unbounded recipient length — while leaving ordinary delimiter/pipe text untouched (arbitrary `<|foo bar=baz|>`, F# `value <| f |> g`, compact `sum<|a+b|>c`, and multi-line bodies never match). The simple branch remains a strict superset of the prior identifier-only pattern (#2267).
15
+
5
16
  ## [0.10.2] - 2026-07-14
6
17
 
7
18
  ### Fixed
@@ -423,7 +434,7 @@
423
434
 
424
435
  - Added `onAuthError` to `StreamOptions` and wired `streamSimple()` to retry once with a replacement API key when the first provider response is a 401 before any assistant events are emitted
425
436
  - Added generation-aware snapshot metadata (`generation`, `serverNowMs`, `refresher`, and `rotatesInMs`) to auth-broker snapshot responses to support client-side credential-rotation planning
426
- - Added `transport: "pi-native"` on `Model` and the matching `streamPiNative` client. When `model.transport === "pi-native"`, `streamSimple` short-circuits the per-provider dispatch and POSTs the canonical `Context` to the auth-gateway's `POST /v1/pi/stream` endpoint. The response is SSE-framed `AssistantMessageEvent`s parsed by `readSseJson` and pushed verbatim into the local `AssistantMessageEventStream` — no wire-format translation, no partial-stripping reconstruction. Used by containerized gjc installs (robogjc slots, swarm extension, etc.) to route every LLM call through a credential-holding sidecar; the slot itself never sees the real provider tokens. Server-controlled fields (`apiKey`, `signal`, `fetch`, lifecycle callbacks, the provider-session map) are stripped from the wire body — `apiKey` rides in the `Authorization` header as the gateway bearer.
437
+ - Added `transport: "pi-native"` on `Model` and the matching `streamPiNative` client. When `model.transport === "pi-native"`, `streamSimple` short-circuits the per-provider dispatch and POSTs the canonical `Context` to the auth-gateway's `POST /v1/pi/stream` endpoint. The response is SSE-framed `AssistantMessageEvent`s parsed by `readSseJson` and pushed verbatim into the local `AssistantMessageEventStream` — no wire-format translation, no partial-stripping reconstruction. Used by containerized GJC deployments and swarm extensions to route every LLM call through a credential-holding sidecar; the container never sees the real provider tokens. Server-controlled fields (`apiKey`, `signal`, `fetch`, lifecycle callbacks, the provider-session map) are stripped from the wire body — `apiKey` rides in the `Authorization` header as the gateway bearer.
427
438
  - Added `POST /v1/pi/stream` to the auth-gateway. Same auth + abort + model-resolution + openai-code-compat + prefix-cache plumbing as the foreign-wire routes; only the wire-format translation is skipped. Request body is `{ modelId, context, options?, stream? }` where `context` is the canonical pi-ai `Context` and `options` is `SimpleStreamOptions` with non-serializable fields stripped. Response is SSE-framed `AssistantMessageEvent` (terminated by `data: [DONE]`) when streaming, or `{ message: AssistantMessage }` JSON when `stream: false`.
428
439
  - Added Vertex AI authentication via Google Application Default Credentials from `GOOGLE_APPLICATION_CREDENTIALS`, `~/.config/gcloud/application_default_credentials.json`, or metadata server tokens, with token caching and refresh skew control via `GOOGLE_VERTEX_REFRESH_SKEW_MS`
429
440
  - Added support for Anthropic image message parts with `type: "url"` and `type: "file"` sources
@@ -442,7 +453,7 @@
442
453
  - Added `AuthStorageOptions.refreshOAuthCredential` override so a remote-store client can route every OAuth refresh through the broker instead of the local OAuth endpoint.
443
454
  - Added `REMOTE_REFRESH_SENTINEL` (`"__remote__"`) — the wire placeholder substituted for OAuth refresh tokens in broker snapshots; clients never see the real refresh token.
444
455
  - Exposed the OAuth provider catalog (`getOAuthProviders`, `OAuthProvider`, `OAuthProviderInfo`) and `refreshOAuthToken` through the package barrel so the coding-agent CLI can target them without reaching into `utils/oauth`.
445
- - Added the auth-gateway subsystem (`@gajae-code/ai/auth-gateway`) — a forward-proxy that sits between unauthenticated clients (the macOS usage widget, llm-git, robogjc containers, …) and the broker. Clients send standard provider-format requests; the gateway parses them into gjc's canonical `Context`, dispatches through pi-ai's `streamSimple()`, and translates the canonical event stream back to the matching wire format. `Authorization` is injected server-side so access tokens never leave the gateway host. Wire surface:
456
+ - Added the auth-gateway subsystem (`@gajae-code/ai/auth-gateway`) — a forward-proxy that sits between unauthenticated clients (the macOS usage widget, llm-git, containerized GJC deployments, …) and the broker. Clients send standard provider-format requests; the gateway parses them into gjc's canonical `Context`, dispatches through pi-ai's `streamSimple()`, and translates the canonical event stream back to the matching wire format. `Authorization` is injected server-side so access tokens never leave the gateway host. Wire surface:
446
457
  - `GET /healthz` — unauth liveness.
447
458
  - `GET /v1/usage` — aggregated provider usage; 5-min per-credential cache via `AuthStorage.fetchUsageReports`.
448
459
  - `GET /v1/models` — model catalog (scoped to providers with credentials).
@@ -1,3 +1,22 @@
1
+ /**
2
+ * gjc auth-gateway HTTP server.
3
+ *
4
+ * Accepts any provider-format request (OpenAI chat-completions, Anthropic
5
+ * messages, OpenAI Responses) and dispatches through pi-ai's `streamSimple()`
6
+ * — which handles credential injection, anthropic-beta headers, OpenAI code backend
7
+ * websocket transport, and all the per-provider intricacies. The gateway is
8
+ * pure protocol translation: foreign wire → gjc Context → pi-ai stream() →
9
+ * gjc events → foreign wire.
10
+ *
11
+ * Endpoints:
12
+ * GET /healthz → unauth; ok + version
13
+ * GET /v1/usage → aggregated provider usage (5-min per-credential cache via AuthStorage)
14
+ * GET /v1/credentials/check → per-credential auth probe (diagnose 401s in a multi-account pool)
15
+ * GET /v1/models → list known models from the registry
16
+ * POST /v1/chat/completions → OpenAI chat-completions in/out
17
+ * POST /v1/messages → Anthropic messages in/out
18
+ * POST /v1/responses → OpenAI Responses in/out
19
+ */
1
20
  import type { AuthStorage } from "../auth-storage";
2
21
  import type { Api, Model } from "../types";
3
22
  import type { AuthGatewayServerHandle, AuthGatewayServerOptions } from "./types";
@@ -510,6 +510,8 @@ export declare class AuthStorage {
510
510
  baseUrlResolver?: (provider: Provider) => string | undefined;
511
511
  /** Caller's cancel signal; only rejects this caller, never the shared upstream fetch. */
512
512
  signal?: AbortSignal;
513
+ /** Disable provider/account/error logging for secret-safe control surfaces. */
514
+ logDetails?: boolean;
513
515
  }): Promise<UsageReport[] | null>;
514
516
  /**
515
517
  * Probe each stored credential against its provider's auth-verifying usage
@@ -0,0 +1,10 @@
1
+ import type { Api, Model } from "./types";
2
+ export interface CodexGpt56ContextCapPolicy {
3
+ fallback: number;
4
+ ceiling: number;
5
+ }
6
+ export declare const CODEX_GPT_5_6_CONTEXT_CAP: CodexGpt56ContextCapPolicy;
7
+ export declare function isCodexProductTransport(model: Pick<Model<Api>, "api" | "provider">): boolean;
8
+ export declare function isCodexGpt56Tier(model: Pick<Model<Api>, "id">): boolean;
9
+ export declare function resolveCodexGpt56DiscoveryContext(model: Pick<Model<Api>, "api" | "id" | "provider">, rawContextWindow: unknown, policy?: CodexGpt56ContextCapPolicy): number;
10
+ export declare function applyFinalCodexGpt56ContextCap<TApi extends Api>(models: readonly Model<TApi>[], policy?: CodexGpt56ContextCapPolicy): Model<TApi>[];
@@ -4,6 +4,7 @@ export * from "./auth-broker";
4
4
  export { type AuthGatewayBootOptions, type ModelResolver, startAuthGateway } from "./auth-gateway/server";
5
5
  export * from "./auth-gateway/types";
6
6
  export * from "./auth-storage";
7
+ export * from "./context-cap-policy";
7
8
  export * from "./model-cache";
8
9
  export * from "./model-manager";
9
10
  export * from "./model-thinking";
@@ -41,6 +42,7 @@ export * from "./usage/zai";
41
42
  export * from "./utils/anthropic-auth";
42
43
  export * from "./utils/discovery";
43
44
  export * from "./utils/event-stream";
45
+ export * from "./utils/fallback-transport";
44
46
  export * from "./utils/h2-fetch";
45
47
  export * from "./utils/oauth";
46
48
  export type { OAuthCredentials, OAuthProvider, OAuthProviderId, OAuthProviderInfo, } from "./utils/oauth/types";
@@ -10,6 +10,7 @@ export type CodexRateLimits = {
10
10
  export type CodexErrorInfo = {
11
11
  message: string;
12
12
  status: number;
13
+ code?: string;
13
14
  friendlyMessage?: string;
14
15
  rateLimits?: CodexRateLimits;
15
16
  raw?: string;
@@ -4,7 +4,7 @@
4
4
  * Where the OpenAI / Anthropic / Responses route modules translate foreign
5
5
  * wire shapes through pi-ai's canonical {@link Context}, this module accepts
6
6
  * the canonical shape *directly* — for clients that already speak pi-ai
7
- * (containerized gjc and robogjc's sidecar auth-gateway).
7
+ * (containerized GJC deployments and sidecar auth gateways).
8
8
  * Skipping the wire-format → Context → wire-format round-trip cuts
9
9
  * per-request CPU but, more importantly, avoids the quantization that those
10
10
  * translations impose on first-class pi-ai fields (service tier, cache
@@ -51,8 +51,8 @@ export declare function parseRequest(body: unknown, _headers?: Headers): PiNativ
51
51
  * canonical event type IS the wire type. Including the rolling
52
52
  * `partial: AssistantMessage` on every delta is quadratic in turn length
53
53
  * on the wire, but for the loopback / sidecar topology this transport
54
- * targets (containerized gjc → host gateway, robogjc slot → gjc-auth-gateway
55
- * sidecar) the bandwidth cost is negligible compared to provider latency —
54
+ * targets (containerized GJC → host gateway) the bandwidth cost is negligible
55
+ * compared to provider latency —
56
56
  * and the client gets to feed the events straight into its existing
57
57
  * `AssistantMessageEventStream.push()` plumbing with zero translation.
58
58
  */
@@ -12,6 +12,7 @@ import type { OpenAICodexResponsesOptions } from "./providers/openai-codex-respo
12
12
  import type { OpenAICompletionsOptions } from "./providers/openai-completions";
13
13
  import type { OpenAIResponsesOptions } from "./providers/openai-responses";
14
14
  import type { AssistantMessageEventStream } from "./utils/event-stream";
15
+ import type { FallbackAttemptToken, TransportFailureFacts } from "./utils/fallback-transport";
15
16
  export type { AssistantMessageEventStream } from "./utils/event-stream";
16
17
  export type KnownApi = "openai-completions" | "openai-responses" | "openai-codex-responses" | "azure-openai-responses" | "anthropic-messages" | "bedrock-converse-stream" | "google-generative-ai" | "google-gemini-cli" | "google-vertex" | "ollama-chat" | "cursor-agent";
17
18
  export type Api = KnownApi | (string & {});
@@ -31,6 +32,8 @@ export interface ApiOptionsMap {
31
32
  export type OptionsForApi<TApi extends Api> = StreamOptions | (TApi extends keyof ApiOptionsMap ? ApiOptionsMap[TApi] : never);
32
33
  /** Canonical thinking transport used by a model. */
33
34
  export type ThinkingControlMode = "effort" | "budget" | "google-level" | "anthropic-adaptive" | "anthropic-budget-effort";
35
+ /** Canonical runtime vocabulary for provider thinking transports. */
36
+ export declare const THINKING_CONTROL_MODES: readonly ["effort", "budget", "google-level", "anthropic-adaptive", "anthropic-budget-effort"];
34
37
  /** Per-model thinking capabilities used to clamp and map user-facing effort levels. */
35
38
  export interface ThinkingConfig {
36
39
  /** Least intensive supported user-facing effort level. */
@@ -163,6 +166,10 @@ export interface StreamOptions {
163
166
  maxTokens?: number;
164
167
  signal?: AbortSignal;
165
168
  apiKey?: string;
169
+ /** Disables all transport-level replay; the fallback controller owns retries. */
170
+ fallbackManaged?: boolean;
171
+ /** Opaque token returned by beginAttempt for a managed transport invocation. */
172
+ fallbackAttempt?: FallbackAttemptToken;
166
173
  /**
167
174
  * Called when a provider returns 401 before any replay-unsafe assistant
168
175
  * event has been emitted. Returning a different key retries the provider
@@ -434,6 +441,8 @@ export interface AssistantMessage {
434
441
  errorKind?: AssistantErrorKind;
435
442
  /** HTTP status surfaced by the provider when the request failed. Populated by every provider's catch block alongside `errorMessage` so consumers (auth retry, telemetry, UI) can branch without regex-scraping the message. */
436
443
  errorStatus?: number;
444
+ /** Typed upstream failure facts retained for retry classification without parsing errorMessage. */
445
+ transportFailure?: TransportFailureFacts;
437
446
  /**
438
447
  * Stable identifiers for request features the provider silently dropped
439
448
  * during this turn (e.g. `"priority"`). Set when a server-side rejection
@@ -790,10 +799,9 @@ export interface Model<TApi extends Api = any> {
790
799
  * (or compatible) host; `headers.Authorization` (or `apiKey` resolved by
791
800
  * the registry) carries the gateway bearer.
792
801
  *
793
- * Used by containerized gjc installs (e.g. robogjc slots) to route every
794
- * LLM call through a sidecar gateway that holds the real provider
795
- * credentials. The model's other metadata (pricing, context window,
796
- * thinking config, …) still resolves locally; only the streaming
802
+ * Used by containerized GJC installs to route every LLM call through a
803
+ * sidecar gateway that holds the real provider credentials. The model's other
804
+ * metadata (pricing, context window, thinking config, …) still resolves locally; only the streaming
797
805
  * dispatch is redirected.
798
806
  */
799
807
  transport?: "pi-native";
@@ -15,11 +15,19 @@ export declare class EventStream<T, R = T> implements AsyncIterable<T> {
15
15
  /**
16
16
  * Read-only snapshot of the not-yet-consumed events. Always a fresh copy:
17
17
  * external code can never mutate internal queue state or observe head-index
18
- * tombstones, so the deque cannot desynchronize.
18
+ * tombstones or private consumer-drain sentinels, so the deque cannot desynchronize.
19
19
  */
20
20
  get queue(): T[];
21
+ /** Read-only test seam for outstanding consumer-drain waiters. */
22
+ get pendingConsumerDrainCountForTests(): number;
21
23
  push(event: T): void;
22
24
  deliver(event: T): void;
25
+ /**
26
+ * Resolves after every event enqueued before this call has been yielded and
27
+ * the consumer asks the iterator for its next node. The private sentinel is
28
+ * never exposed through the async iterator.
29
+ */
30
+ waitForConsumerDrain(signal: AbortSignal): Promise<void>;
23
31
  end(result?: R): void;
24
32
  endWaiting(): void;
25
33
  fail(err: unknown): void;
@@ -28,6 +36,4 @@ export declare class EventStream<T, R = T> implements AsyncIterable<T> {
28
36
  }
29
37
  export declare class AssistantMessageEventStream extends EventStream<AssistantMessageEvent, AssistantMessage> {
30
38
  constructor();
31
- push(event: AssistantMessageEvent): void;
32
- end(result?: AssistantMessage): void;
33
39
  }
@@ -0,0 +1,55 @@
1
+ export type FallbackTriggerClass = "rate_limit" | "quota" | "auth" | "server" | "other";
2
+ export interface FallbackTrigger {
3
+ class: FallbackTriggerClass;
4
+ retryAfterMs?: number;
5
+ }
6
+ export type TransportHeaders = Headers | Record<string, string | undefined>;
7
+ /**
8
+ * Structured facts from an upstream HTTP or transport failure. Retry decisions
9
+ * must use these facts rather than provider- or application-owned error text.
10
+ */
11
+ export interface TransportFailureFacts {
12
+ kind: "transport";
13
+ status?: number;
14
+ providerCode?: string;
15
+ headers?: TransportHeaders;
16
+ }
17
+ /** Opaque per-invocation marker required by managed fallback transport calls. */
18
+ export interface FallbackAttemptToken {
19
+ readonly modelKey: string;
20
+ readonly attemptId: string | number;
21
+ }
22
+ /**
23
+ * Marks a single outer fallback invocation. Accounting belongs to the caller;
24
+ * this token prevents managed transport calls from silently bypassing it.
25
+ */
26
+ export declare function beginAttempt(modelKey: string, attemptId: string | number): FallbackAttemptToken;
27
+ export declare function assertManagedAttempt(options: {
28
+ fallbackManaged?: boolean;
29
+ fallbackAttempt?: FallbackAttemptToken;
30
+ } | undefined): void;
31
+ /**
32
+ * Compatibility input for callers that have not yet wrapped their HTTP facts
33
+ * in the discriminated form. Only its structured fields are inspected.
34
+ */
35
+ export interface FallbackTriggerInput {
36
+ status?: number;
37
+ providerCode?: string;
38
+ code?: string;
39
+ headers?: TransportHeaders;
40
+ response?: {
41
+ status?: number;
42
+ headers?: TransportHeaders;
43
+ };
44
+ error?: {
45
+ code?: string;
46
+ type?: string;
47
+ };
48
+ }
49
+ /** Extracts only explicit HTTP/transport metadata; it never parses error text. */
50
+ export declare function transportFailureFacts(error: unknown, capturedResponse?: {
51
+ status?: number;
52
+ headers?: TransportHeaders;
53
+ }): TransportFailureFacts | undefined;
54
+ /** Classifies only typed upstream transport facts without consuming response bodies. */
55
+ export declare function classifyFallbackTrigger(errorOrFacts: TransportFailureFacts | FallbackTriggerInput | unknown): FallbackTrigger;
@@ -23,4 +23,5 @@ export declare function callWithCopilotModelRetry<T>(fn: () => Promise<T>, optio
23
23
  provider: string;
24
24
  signal?: AbortSignal;
25
25
  retryBaseDelayMs?: number;
26
+ fallbackManaged?: boolean;
26
27
  }): Promise<T>;
@@ -28,6 +28,23 @@ export declare function sanitizeOpenAIResponsesHistoryItemsForReplay(items: Arra
28
28
  * the offending item is re-sent on each turn. Insert a zero-width space after `<`
29
29
  * so the delimiter can no longer be tokenized as a reserved control token while the
30
30
  * text stays human-readable.
31
+ *
32
+ * The pattern matches the two control-token shapes only, so ordinary text and pipe
33
+ * syntax is left untouched:
34
+ * - simple form `<|ident|>` — a leading run of identifier chars then `|>`; and
35
+ * - header form `<|role to=recipient|>` — a known Harmony role
36
+ * (`system`/`developer`/`user`/`assistant`/`tool`) followed by a single
37
+ * recipient assignment `to=<recipient>` whose value is an unbounded run of
38
+ * non-delimiter, non-whitespace chars (so long MCP/custom tool recipients like
39
+ * `to=functions.<long.name>` are covered).
40
+ * The header branch is deliberately scoped to the known role + `to=` recipient
41
+ * grammar rather than an arbitrary `key=value`, so request-boundary sanitization
42
+ * never rewrites non-control delimiter text such as `<|foo bar=baz|>`. A single-line
43
+ * body (no `\n`) and the required leading identifier char also leave compact
44
+ * pipe/operator syntax alone — e.g. F# `value <| f |> g` (space after `<|`),
45
+ * `sum<|a+b|>c` (punctuation body), and `<|foo bar|>` (no assignment) never match.
46
+ * The simple branch is a strict superset of the original identifier-only pattern:
47
+ * every marker the old regex caught still matches.
31
48
  */
32
49
  export declare function neutralizeReservedControlTokens(text: string): string;
33
50
  /**
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.10.2",
4
+ "version": "0.11.0",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -40,7 +40,7 @@
40
40
  "dependencies": {
41
41
  "@anthropic-ai/sdk": "^0.94.0",
42
42
  "@bufbuild/protobuf": "^2.12.0",
43
- "@gajae-code/utils": "0.10.2",
43
+ "@gajae-code/utils": "0.11.0",
44
44
  "openai": "^6.36.0",
45
45
  "partial-json": "^0.1.7",
46
46
  "zod": "4.4.3"
@@ -17,6 +17,7 @@
17
17
  * POST /v1/messages → Anthropic messages in/out
18
18
  * POST /v1/responses → OpenAI Responses in/out
19
19
  */
20
+
20
21
  import { logger } from "@gajae-code/utils";
21
22
  import type { AuthStorage } from "../auth-storage";
22
23
  import { Effort } from "../model-thinking";
@@ -26,6 +27,7 @@ import * as openaiResponses from "../providers/openai-responses-server";
26
27
  import * as piNative from "../providers/pi-native-server";
27
28
  import { streamSimple } from "../stream";
28
29
  import type { Api, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types";
30
+ import { beginAttempt, classifyFallbackTrigger } from "../utils/fallback-transport";
29
31
  import { parseBind } from "../utils/parse-bind";
30
32
  import {
31
33
  captureRequestHeaders,
@@ -260,6 +262,90 @@ async function refreshGatewayApiKeyAfterAuthError(
260
262
  return storage.getApiKey(provider, undefined, { modelId: model.id, signal });
261
263
  }
262
264
 
265
+ /**
266
+ * Records a managed gateway failure against the credential selected for this
267
+ * request. This deliberately never returns a replacement key: the outer
268
+ * fallback controller owns the next attempt and is the only component allowed
269
+ * to make another upstream request.
270
+ */
271
+ async function markManagedGatewayCredentialFailure(
272
+ storage: AuthStorage,
273
+ model: Model<Api>,
274
+ apiKey: string,
275
+ error: unknown,
276
+ signal: AbortSignal,
277
+ format: string,
278
+ peer: string,
279
+ ): Promise<void> {
280
+ const trigger = classifyFallbackTrigger(error);
281
+ try {
282
+ if (trigger.class === "auth") {
283
+ await storage.invalidateCredentialMatching(model.provider, apiKey, signal);
284
+ } else if (trigger.class === "quota" || trigger.class === "rate_limit") {
285
+ await storage.markUsageLimitReached(model.provider, undefined, {
286
+ retryAfterMs: trigger.retryAfterMs,
287
+ signal,
288
+ });
289
+ } else {
290
+ return;
291
+ }
292
+ logger.debug("auth-gateway recorded managed credential failure", {
293
+ format,
294
+ provider: model.provider,
295
+ peer,
296
+ trigger: trigger.class,
297
+ });
298
+ } catch (markError) {
299
+ // Credential bookkeeping must not replace the upstream failure returned to
300
+ // the fallback controller.
301
+ logger.warn("auth-gateway failed to record managed credential failure", {
302
+ format,
303
+ provider: model.provider,
304
+ peer,
305
+ error: markError instanceof Error ? markError.message : String(markError),
306
+ });
307
+ }
308
+ }
309
+
310
+ function observeManagedGatewayFailure(
311
+ events: AssistantMessageEventStream,
312
+ markFailure: (error: unknown) => Promise<void>,
313
+ ): AssistantMessageEventStream {
314
+ let marked = false;
315
+ const markOnce = async (error: unknown): Promise<void> => {
316
+ if (marked) return;
317
+ marked = true;
318
+ await markFailure(error);
319
+ };
320
+ async function* observed() {
321
+ try {
322
+ for await (const event of events) {
323
+ if (event.type === "error") {
324
+ await markOnce(event.error.transportFailure ?? { kind: "transport", status: event.error.errorStatus });
325
+ }
326
+ yield event;
327
+ }
328
+ } catch (error) {
329
+ await markOnce(error);
330
+ throw error;
331
+ }
332
+ }
333
+ const result = observed() as unknown as AssistantMessageEventStream;
334
+ result.result = async () => {
335
+ try {
336
+ const message = await events.result();
337
+ if (message.stopReason === "error") {
338
+ await markOnce(message.transportFailure ?? { kind: "transport", status: message.errorStatus });
339
+ }
340
+ return message;
341
+ } catch (error) {
342
+ await markOnce(error);
343
+ throw error;
344
+ }
345
+ };
346
+ return result;
347
+ }
348
+
263
349
  function clientClosedResponse(route: { module: FormatModule }): Response {
264
350
  return route.module.formatError(499, "request_aborted", "client closed request");
265
351
  }
@@ -361,17 +447,21 @@ async function handleFormatEndpoint(
361
447
 
362
448
  const streamOpts = buildStreamOptions(parsed, model.api, controller.signal);
363
449
  streamOpts.apiKey = apiKey;
364
- streamOpts.onAuthError = (provider, oldKey, error) =>
365
- refreshGatewayApiKeyAfterAuthError(
366
- bootOpts.storage,
367
- model,
368
- provider,
369
- oldKey,
370
- error,
371
- controller.signal,
372
- route.label,
373
- peer,
374
- );
450
+ if (streamOpts.fallbackManaged) {
451
+ streamOpts.fallbackAttempt = beginAttempt(model.id, "auth-gateway");
452
+ } else {
453
+ streamOpts.onAuthError = (provider, oldKey, error) =>
454
+ refreshGatewayApiKeyAfterAuthError(
455
+ bootOpts.storage,
456
+ model,
457
+ provider,
458
+ oldKey,
459
+ error,
460
+ controller.signal,
461
+ route.label,
462
+ peer,
463
+ );
464
+ }
375
465
 
376
466
  logger.info("auth-gateway request", {
377
467
  format: route.label,
@@ -387,10 +477,34 @@ async function handleFormatEndpoint(
387
477
  if (controller.signal.aborted) return clientClosedResponse(route);
388
478
  events = streamSimple(model, parsed.context, streamOpts);
389
479
  } catch (error) {
480
+ if (streamOpts.fallbackManaged) {
481
+ await markManagedGatewayCredentialFailure(
482
+ bootOpts.storage,
483
+ model,
484
+ apiKey,
485
+ error,
486
+ controller.signal,
487
+ route.label,
488
+ peer,
489
+ );
490
+ }
390
491
  const classified = classifyGatewayError(error);
391
492
  logger.warn("auth-gateway streamSimple threw", { format: route.label, error: classified.message, peer });
392
493
  return route.module.formatError(classified.status, classified.type, classified.message);
393
494
  }
495
+ if (streamOpts.fallbackManaged) {
496
+ events = observeManagedGatewayFailure(events, error =>
497
+ markManagedGatewayCredentialFailure(
498
+ bootOpts.storage,
499
+ model,
500
+ apiKey,
501
+ error,
502
+ controller.signal,
503
+ route.label,
504
+ peer,
505
+ ),
506
+ );
507
+ }
394
508
 
395
509
  if (!parsed.stream) {
396
510
  try {
@@ -509,17 +623,21 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe
509
623
  // only inject server-controlled fields. The OpenAI code backend temperature/topP strip
510
624
  // matches `buildStreamOptions` — OpenAI code backend rejects them with a 400.
511
625
  const streamOpts: SimpleStreamOptions = { ...parsed.options, apiKey, signal: controller.signal };
512
- streamOpts.onAuthError = (provider, oldKey, error) =>
513
- refreshGatewayApiKeyAfterAuthError(
514
- bootOpts.storage,
515
- model,
516
- provider,
517
- oldKey,
518
- error,
519
- controller.signal,
520
- "pi-native",
521
- peer,
522
- );
626
+ if (streamOpts.fallbackManaged) {
627
+ streamOpts.fallbackAttempt = beginAttempt(model.id, "auth-gateway-pi-native");
628
+ } else {
629
+ streamOpts.onAuthError = (provider, oldKey, error) =>
630
+ refreshGatewayApiKeyAfterAuthError(
631
+ bootOpts.storage,
632
+ model,
633
+ provider,
634
+ oldKey,
635
+ error,
636
+ controller.signal,
637
+ "pi-native",
638
+ peer,
639
+ );
640
+ }
523
641
  if (model.api === "openai-codex-responses") {
524
642
  delete streamOpts.temperature;
525
643
  delete streamOpts.topP;
@@ -547,10 +665,34 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe
547
665
  if (controller.signal.aborted) return aborted();
548
666
  events = streamSimple(model, parsed.context, streamOpts);
549
667
  } catch (error) {
668
+ if (streamOpts.fallbackManaged) {
669
+ await markManagedGatewayCredentialFailure(
670
+ bootOpts.storage,
671
+ model,
672
+ apiKey,
673
+ error,
674
+ controller.signal,
675
+ "pi-native",
676
+ peer,
677
+ );
678
+ }
550
679
  const classified = classifyGatewayError(error);
551
680
  logger.warn("auth-gateway streamSimple threw", { format: "pi-native", error: classified.message, peer });
552
681
  return piNative.formatError(classified.status, classified.type, classified.message);
553
682
  }
683
+ if (streamOpts.fallbackManaged) {
684
+ events = observeManagedGatewayFailure(events, error =>
685
+ markManagedGatewayCredentialFailure(
686
+ bootOpts.storage,
687
+ model,
688
+ apiKey,
689
+ error,
690
+ controller.signal,
691
+ "pi-native",
692
+ peer,
693
+ ),
694
+ );
695
+ }
554
696
 
555
697
  if (!parsed.stream) {
556
698
  try {