@gajae-code/ai 0.11.0 → 0.11.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,12 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.11.2] - 2026-07-19
6
+
7
+ ### Fixed
8
+
9
+ - `transportFailureFacts` now reduces transport headers to a plain record containing only the retained retry signals (`retry-after`, `retry-after-ms`). Providers attach these facts to error `AssistantMessage`s, and the previous shape carried the live fetch/SDK `Headers` instance — which is not structured-cloneable (`structuredClone` throws `DataCloneError`, "The object can not be cloned." under Bun) and not JSON-serializable (persisted as `{}` in session files, silently dropping the retry hint). Under a managed model fallback chain, snapshotting such an error message replaced the real provider failure with the local clone error and exhausted the whole chain. Normalization is idempotent (re-running facts on facts is structurally stable; errors carrying only unretained headers with no status/code now yield no facts instead of an empty facts object), Retry-After classification (`classifyFallbackTrigger`) is unchanged, and arbitrary response headers no longer reach persisted facts.
10
+
5
11
  ## [0.11.0] - 2026-07-15
6
12
  ### Added
7
13
 
@@ -13,6 +19,8 @@
13
19
 
14
20
  - Closed the remaining `Request blocked (code=invalid_prompt)` wedge on gpt-5.6 caused by header-form leaked Harmony markers. The reserved-control-token sanitizer only matched the simple `<|ident|>` shape, so a header-form marker carrying a recipient (e.g. `<|assistant to=functions.bash|>`) survived every sanitizer path (replay, request boundary, compaction) and kept re-poisoning history even after the earlier fixes. The pattern now also matches the scoped header grammar — a known Harmony role (`system`/`developer`/`user`/`assistant`/`tool`) plus a `to=<recipient>` assignment with unbounded recipient length — while leaving ordinary delimiter/pipe text untouched (arbitrary `<|foo bar=baz|>`, F# `value <| f |> g`, compact `sum<|a+b|>c`, and multi-line bodies never match). The simple branch remains a strict superset of the prior identifier-only pattern (#2267).
15
21
 
22
+ - Made the `Request blocked (code=invalid_prompt)` classification explicit and shared across transports (#2282). `invalid_prompt` was only non-retryable by omission — it appeared in neither the codex retryable nor non-retryable event set, and the plain OpenAI Responses transport surfaced it as a generic error with no durable marker. It is now in the codex `CODEX_NON_RETRYABLE_EVENT_CODES` set (code and message forms), the Responses error path tags `transportFailure.providerCode = "invalid_prompt"`, and a new exported `isInvalidPromptError` predicate is the single contract both transports and the session-level circuit breaker key on. Ordinary control-token / pipe text (F# `value <| f |> g`, `sum<|a+b|>c`, `<|foo bar=baz|>`) is unaffected; genuinely transient errors (`server_error`, `model_error`) stay retryable.
23
+
16
24
  ## [0.10.2] - 2026-07-14
17
25
 
18
26
  ### Fixed
@@ -410,6 +410,8 @@ export declare class AuthStorage {
410
410
  * Remove a runtime API key override.
411
411
  */
412
412
  removeRuntimeApiKey(provider: string): void;
413
+ /** Whether a provider is currently authenticated by a runtime API-key override. */
414
+ hasRuntimeApiKey(provider: string): boolean;
413
415
  /**
414
416
  * Register a per-provider API key sourced from user configuration
415
417
  * (e.g. `models.yml` `providers.<name>.apiKey`). Higher priority than
@@ -12,6 +12,8 @@ interface CacheEntry<TApi extends Api = Api> {
12
12
  */
13
13
  staticFingerprint: string;
14
14
  }
15
+ /** Close the shared cache only when it owns the exact requested database path. */
16
+ export declare function closeModelCache(dbPath?: string): boolean;
15
17
  export declare function readModelCache<TApi extends Api>(providerId: string, ttlMs: number, now: () => number, dbPath?: string): CacheEntry<TApi> | null;
16
18
  export declare function writeModelCache<TApi extends Api>(providerId: string, updatedAt: number, models: Model<TApi>[], authoritative: boolean, staticFingerprint: string, dbPath?: string): void;
17
19
  export {};
@@ -30,6 +30,8 @@ export interface ModelManagerOptions<TApi extends Api = Api, TModelsDevPayload =
30
30
  modelsDev?: ModelsDevFallback<TApi, TModelsDevPayload>;
31
31
  /** Clock override for deterministic tests. */
32
32
  now?: () => number;
33
+ /** Optional guard that must permit cache publication. Default: writes are permitted. */
34
+ canPublishCache?: () => boolean;
33
35
  }
34
36
  /**
35
37
  * Resolution result.
@@ -1,5 +1,5 @@
1
1
  import Anthropic, { type ClientOptions as AnthropicSdkClientOptions } from "@anthropic-ai/sdk";
2
- import type { MessageParam } from "@anthropic-ai/sdk/resources/messages";
2
+ import type { MessageCreateParamsStreaming, MessageParam } from "@anthropic-ai/sdk/resources/messages";
3
3
  import type { FetchImpl, Message, Model, ProviderSessionState, ServiceTier, SimpleStreamOptions, StreamFunction, StreamOptions, Usage } from "../types";
4
4
  export type AnthropicHeaderOptions = {
5
5
  apiKey: string;
@@ -181,6 +181,7 @@ type SystemBlockOptions = {
181
181
  export declare function buildAnthropicSystemBlocks(systemPrompt: readonly string[] | undefined, options?: SystemBlockOptions): AnthropicSystemBlock[] | undefined;
182
182
  export declare function normalizeExtraBetas(betas?: string[] | string): string[];
183
183
  export declare function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): AnthropicClientOptionsResult;
184
+ export declare function normalizeCacheControlTtlOrdering(params: MessageCreateParamsStreaming): void;
184
185
  export declare function convertAnthropicMessages(messages: Message[], model: Model<"anthropic-messages">, isOAuthToken: boolean, options?: {
185
186
  repairLatestAssistantThinking?: boolean;
186
187
  }): MessageParam[];
@@ -45,8 +45,8 @@ export declare function getOpenAICodexTransportDetails(model: Model<"openai-code
45
45
  providerSessionState?: Map<string, ProviderSessionState>;
46
46
  }): OpenAICodexTransportDetails;
47
47
  declare function convertMessages(model: Model<"openai-codex-responses">, context: Context): ResponseInput;
48
- /** @internal Exported for tests. */
49
- export { convertMessages as convertCodexResponsesMessages };
48
+ /** @internal Exported for tests. `classifyCodexFailureEventRetryable` is the retry classification of a Codex failure event. */
49
+ export { convertMessages as convertCodexResponsesMessages, isRetryableCodexFailureEvent as classifyCodexFailureEventRetryable, };
50
50
  type CodexToolPayload = {
51
51
  type: "function";
52
52
  name: string;
@@ -65,3 +65,4 @@ type CodexToolPayload = {
65
65
  };
66
66
  /** @internal Exported for tests. */
67
67
  export declare function convertOpenAICodexResponsesTools(tools: Tool[], model: Model<"openai-codex-responses">): CodexToolPayload[];
68
+ declare function isRetryableCodexFailureEvent(rawEvent: Record<string, unknown>): boolean;
@@ -10,13 +10,12 @@
10
10
  * translations impose on first-class pi-ai fields (service tier, cache
11
11
  * markers, thinking budgets, tool-choice variants, …).
12
12
  *
13
- * The streaming wire is {@link AssistantMessageEvent} serialized verbatim and
14
- * SSE-framed. Same type pi-ai already produces internally; the client feeds
15
- * each parsed event straight into `AssistantMessageEventStream.push()` with
16
- * no translation. Including `partial: AssistantMessage` on every delta is
17
- * O(N²) in turn length on the wire — acceptable for the loopback / sidecar
18
- * topology this transport is designed for; provider latency dominates the
19
- * actual cost.
13
+ * The streaming wire is {@link AssistantMessageEvent} serialized as SSE. Public
14
+ * projections omit private raw reasoning and serialized Responses reasoning
15
+ * signatures while preserving provider-displayable summaries and genuine opaque
16
+ * signatures. Including `partial: AssistantMessage` on every delta is O(N²) in
17
+ * turn length on the wire — acceptable for the loopback / sidecar topology this
18
+ * transport is designed for; provider latency dominates the actual cost.
20
19
  *
21
20
  * Endpoint contract:
22
21
  * POST /v1/pi/stream
@@ -45,16 +44,9 @@ export interface PiNativeParsedRequest {
45
44
  */
46
45
  export declare function parseRequest(body: unknown, _headers?: Headers): PiNativeParsedRequest;
47
46
  /**
48
- * Ship every {@link AssistantMessageEvent} verbatim, SSE-framed.
49
- *
50
- * No per-event re-shaping: the pi-native client is pi-ai itself, so the
51
- * canonical event type IS the wire type. Including the rolling
52
- * `partial: AssistantMessage` on every delta is quadratic in turn length
53
- * on the wire, but for the loopback / sidecar topology this transport
54
- * targets (containerized GJC → host gateway) the bandwidth cost is negligible
55
- * compared to provider latency —
56
- * and the client gets to feed the events straight into its existing
57
- * `AssistantMessageEventStream.push()` plumbing with zero translation.
47
+ * Ship only public-safe {@link AssistantMessageEvent} projections. Unknown
48
+ * thinking blocks remain buffered until their terminal partial establishes that
49
+ * the provider-native block is safe; raw and mixed blocks never reach SSE.
58
50
  */
59
51
  export declare function encodeStream(events: AssistantMessageEventStream): ReadableStream<Uint8Array>;
60
52
  /**
@@ -319,6 +319,9 @@ export interface ThinkingContent {
319
319
  thinking: string;
320
320
  thinkingSignature?: string;
321
321
  itemId?: string;
322
+ readonly provenance?: "summary" | "raw" | "mixed";
323
+ readonly summaryText?: string;
324
+ readonly rawText?: string;
322
325
  }
323
326
  export interface RedactedThinkingContent {
324
327
  type: "redactedThinking";
@@ -542,6 +545,16 @@ export interface Tool<TParameters extends TSchema = TSchema> {
542
545
  * calls route correctly. Absent for regular JSON function tools.
543
546
  */
544
547
  customWireName?: string;
548
+ /**
549
+ * Optional safe projection for tool arguments or results. Extensions use this
550
+ * only for explicitly opt-in, display-safe summaries.
551
+ */
552
+ safeSummary?: (kind: "args" | "result", value: unknown) => string | undefined;
553
+ /** Allowlisted argument/result field names for a safe fallback summary. */
554
+ safeSummaryFields?: {
555
+ args?: string[];
556
+ result?: string[];
557
+ };
545
558
  }
546
559
  export interface Context {
547
560
  systemPrompt?: string[];
@@ -580,6 +593,20 @@ export type AssistantMessageEvent = {
580
593
  contentIndex: number;
581
594
  content: string;
582
595
  partial: AssistantMessage;
596
+ } | {
597
+ type: "reasoning_summary_start";
598
+ contentIndex: number;
599
+ partial: AssistantMessage;
600
+ } | {
601
+ type: "reasoning_summary_delta";
602
+ contentIndex: number;
603
+ delta: string;
604
+ partial: AssistantMessage;
605
+ } | {
606
+ type: "reasoning_summary_end";
607
+ contentIndex: number;
608
+ content: string;
609
+ partial: AssistantMessage;
583
610
  } | {
584
611
  type: "toolcall_start";
585
612
  contentIndex: number;
@@ -731,6 +758,12 @@ export interface AnthropicCompat extends ToolChoiceCompat {
731
758
  supportsForcedToolChoice?: boolean;
732
759
  /** Whether long prompt-cache retention (`ttl: "1h"`) is supported. Default: true for canonical Anthropic API. */
733
760
  supportsLongCacheRetention?: boolean;
761
+ /**
762
+ * Prompt-cache transport accepted by this Anthropic-compatible endpoint.
763
+ * Canonical Anthropic defaults to `"automatic"`; noncanonical endpoints default
764
+ * to `"none"` and must explicitly opt into generated `"explicit"` markers.
765
+ */
766
+ promptCacheMode?: "none" | "explicit" | "automatic";
734
767
  }
735
768
  /**
736
769
  * OpenRouter provider routing preferences.
@@ -1,4 +1,4 @@
1
- export type FallbackTriggerClass = "rate_limit" | "quota" | "auth" | "server" | "other";
1
+ export type FallbackTriggerClass = "rate_limit" | "quota" | "auth" | "server" | "unknown" | "other";
2
2
  export interface FallbackTrigger {
3
3
  class: FallbackTriggerClass;
4
4
  retryAfterMs?: number;
@@ -7,12 +7,23 @@ export type TransportHeaders = Headers | Record<string, string | undefined>;
7
7
  /**
8
8
  * Structured facts from an upstream HTTP or transport failure. Retry decisions
9
9
  * must use these facts rather than provider- or application-owned error text.
10
+ *
11
+ * `headers` is always a plain record limited to the retained retry-signal
12
+ * entries: facts travel on persisted `AssistantMessage`s and through
13
+ * `structuredClone` snapshots (managed fallback attempt staging), so they must
14
+ * never carry a live `Headers` instance — cloning one throws `DataCloneError`
15
+ * ("The object can not be cloned.") and masks the real provider failure.
10
16
  */
11
17
  export interface TransportFailureFacts {
12
18
  kind: "transport";
13
19
  status?: number;
20
+ /** Canonical provider error code used for fallback classification. */
14
21
  providerCode?: string;
15
- headers?: TransportHeaders;
22
+ /** Anthropic's typed `error.type`, preserved separately at the transport boundary. */
23
+ anthropicErrorType?: string;
24
+ /** OpenAI's typed `error.code`, preserved separately at the transport boundary. */
25
+ openaiErrorCode?: string;
26
+ headers?: Record<string, string>;
16
27
  }
17
28
  /** Opaque per-invocation marker required by managed fallback transport calls. */
18
29
  export interface FallbackAttemptToken {
@@ -1,61 +1,11 @@
1
1
  import type { AssistantMessage } from "../types";
2
+ import type { TransportFailureFacts } from "./fallback-transport";
3
+ export declare function classifyContextOverflow(message: AssistantMessage, transportFailure?: TransportFailureFacts, contextWindow?: number): boolean;
2
4
  /**
3
5
  * Check if an assistant message represents a context overflow error.
4
6
  *
5
- * This handles three cases:
6
- * 1. Error-based overflow: Most providers return stopReason "error" with a
7
- * specific error message pattern.
8
- * 2. Silent overflow: Some providers accept overflow requests and return
9
- * successfully. For these, we check if usage.input exceeds the context window.
10
- * 3. Proxy-level overflow: Some proxies (e.g. LiteLLM) return a "successful"
11
- * response with empty content and a fabricated near-zero usage when the
12
- * upstream model's context window is exceeded.
13
- *
14
- * ## Reliability by Provider
15
- *
16
- * **Reliable detection (returns error with detectable message):**
17
- * - Anthropic: "prompt is too long: X tokens > Y maximum"
18
- * - OpenAI (Completions & Responses): "exceeds the context window"
19
- * - Google Gemini: "input token count exceeds the maximum"
20
- * - xAI (Grok): "maximum prompt length is X but request contains Y"
21
- * - Groq: "reduce the length of the messages"
22
- * - Cerebras: 400/413 status code (no body)
23
- * - Mistral: 400/413 status code (no body)
24
- * - HTTP 413 payload/entity-too-large variants
25
- * - OpenRouter (all backends): "maximum context length is X tokens"
26
- * - llama.cpp: "exceeds the available context size"
27
- * - LM Studio: "greater than the context length"
28
- * - Kimi For Coding: "exceeded model token limit: X (requested: Y)"
29
- * - Anthropic 413: "request_too_large" (request body exceeds size limit)
30
- * - HTTP 413: "Payload Too Large" / "Request Entity Too Large"
31
- *
32
- * **Unreliable detection:**
33
- * - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
34
- * sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.
35
- * - Ollama: Silently truncates input without error. Cannot be detected via this function.
36
- * - LiteLLM proxy: Returns a "successful" response with empty content and a
37
- * fabricated near-zero usage (e.g. input: 1, output: 1) when the upstream
38
- * model's context window is exceeded. Detected via Case 3 (empty content +
39
- * anomalously low usage). Note: the LiteLLM proxy's context limit may differ
40
- * from the underlying model's advertised contextWindow (e.g. configured via
41
- * `model_info.max_tokens` in LiteLLM's config.yaml), so Case 2 (which compares
42
- * usage.input against contextWindow) may not catch it.
43
- * The response will have usage.input < expected, but we don't know the expected value.
44
- *
45
- * ## Custom Providers
46
- *
47
- * If you've added custom models via settings.json, this function may not detect
48
- * overflow errors from those providers. To add support:
49
- *
50
- * 1. Send a request that exceeds the model's context window
51
- * 2. Check the errorMessage in the response
52
- * 3. Create a regex pattern that matches the error
53
- * 4. The pattern should be added to OVERFLOW_PATTERNS in this file, or
54
- * check the errorMessage yourself before calling this function
55
- *
56
- * @param message - The assistant message to check
57
- * @param contextWindow - Optional context window size for detecting silent overflow (z.ai)
58
- * @returns true if the message indicates a context overflow
7
+ * Callers with normalized transport facts should use {@link classifyContextOverflow}
8
+ * so typed provider codes take precedence over error prose.
59
9
  */
60
10
  export declare function isContextOverflow(message: AssistantMessage, contextWindow?: number): boolean;
61
11
  /**
@@ -47,6 +47,18 @@ export declare function sanitizeOpenAIResponsesHistoryItemsForReplay(items: Arra
47
47
  * every marker the old regex caught still matches.
48
48
  */
49
49
  export declare function neutralizeReservedControlTokens(text: string): string;
50
+ /**
51
+ * Shape-tolerant classifier for the poisoned-history rejection that wedges
52
+ * gpt-5.6 sessions: `Request blocked (code=invalid_prompt)`. Accepts a raw
53
+ * provider error, an assistant message, or any object carrying a
54
+ * `providerCode` / `transportFailure` / `errorMessage` field, and returns true
55
+ * when the failure is the deterministic `invalid_prompt` content fault rather
56
+ * than a transient upstream error. This is the single shared contract the
57
+ * provider transports and the session-level circuit breaker key on so the
58
+ * classification is explicit (not inferred from a catch-all bucket) and
59
+ * uniformly testable across transports.
60
+ */
61
+ export declare function isInvalidPromptError(input: unknown): boolean;
50
62
  /**
51
63
  * Neutralize leaked reserved control tokens across every string in an outgoing
52
64
  * Responses `input` array. This is the request-boundary complement to the
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.11.0",
4
+ "version": "0.11.2",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -40,7 +40,7 @@
40
40
  "dependencies": {
41
41
  "@anthropic-ai/sdk": "^0.94.0",
42
42
  "@bufbuild/protobuf": "^2.12.0",
43
- "@gajae-code/utils": "0.11.0",
43
+ "@gajae-code/utils": "0.11.2",
44
44
  "openai": "^6.36.0",
45
45
  "partial-json": "^0.1.7",
46
46
  "zod": "4.4.3"
@@ -775,7 +775,9 @@ export class AuthStorage {
775
775
  */
776
776
  static async create(dbPath: string, options: AuthStorageOptions = {}): Promise<AuthStorage> {
777
777
  const store = await SqliteAuthCredentialStore.open(dbPath);
778
- return new AuthStorage(store, options);
778
+ const storage = new AuthStorage(store, options);
779
+ await storage.reload();
780
+ return storage;
779
781
  }
780
782
 
781
783
  /**
@@ -854,6 +856,7 @@ export class AuthStorage {
854
856
  */
855
857
  setRuntimeApiKey(provider: string, apiKey: string): void {
856
858
  this.#runtimeOverrides.set(provider, apiKey);
859
+ this.#bumpGeneration("set-runtime-api-key");
857
860
  }
858
861
 
859
862
  /**
@@ -877,7 +880,12 @@ export class AuthStorage {
877
880
  * Remove a runtime API key override.
878
881
  */
879
882
  removeRuntimeApiKey(provider: string): void {
880
- this.#runtimeOverrides.delete(provider);
883
+ if (this.#runtimeOverrides.delete(provider)) this.#bumpGeneration("remove-runtime-api-key");
884
+ }
885
+
886
+ /** Whether a provider is currently authenticated by a runtime API-key override. */
887
+ hasRuntimeApiKey(provider: string): boolean {
888
+ return Boolean(this.#runtimeOverrides.get(provider));
881
889
  }
882
890
 
883
891
  /**
@@ -892,13 +900,14 @@ export class AuthStorage {
892
900
  */
893
901
  setConfigApiKey(provider: string, apiKey: string): void {
894
902
  this.#configOverrides.set(provider, apiKey);
903
+ this.#bumpGeneration("set-config-api-key");
895
904
  }
896
905
 
897
906
  /**
898
907
  * Remove a single config-sourced API key override.
899
908
  */
900
909
  removeConfigApiKey(provider: string): void {
901
- this.#configOverrides.delete(provider);
910
+ if (this.#configOverrides.delete(provider)) this.#bumpGeneration("remove-config-api-key");
902
911
  }
903
912
 
904
913
  /**
@@ -906,7 +915,9 @@ export class AuthStorage {
906
915
  * re-parsing `models.yml` so removed entries actually disappear.
907
916
  */
908
917
  clearConfigApiKeys(): void {
918
+ if (this.#configOverrides.size === 0) return;
909
919
  this.#configOverrides.clear();
920
+ this.#bumpGeneration("clear-config-api-keys");
910
921
  }
911
922
 
912
923
  /**
@@ -66,6 +66,16 @@ function getDb(dbPath?: string): Database {
66
66
  return db;
67
67
  }
68
68
 
69
+ /** Close the shared cache only when it owns the exact requested database path. */
70
+ export function closeModelCache(dbPath?: string): boolean {
71
+ const resolvedPath = dbPath ?? getModelDbPath();
72
+ if (!sharedDb || sharedDbPath !== resolvedPath) return false;
73
+ sharedDb.close();
74
+ sharedDb = null;
75
+ sharedDbPath = null;
76
+ return true;
77
+ }
78
+
69
79
  function migrateCacheSchema(db: Database): void {
70
80
  const columns = db.prepare("PRAGMA table_info(model_cache)").all() as TableInfoRow[];
71
81
  if (!columns.some(column => column.name === "static_fingerprint")) {
@@ -41,6 +41,8 @@ export interface ModelManagerOptions<TApi extends Api = Api, TModelsDevPayload =
41
41
  modelsDev?: ModelsDevFallback<TApi, TModelsDevPayload>;
42
42
  /** Clock override for deterministic tests. */
43
43
  now?: () => number;
44
+ /** Optional guard that must permit cache publication. Default: writes are permitted. */
45
+ canPublishCache?: () => boolean;
44
46
  }
45
47
 
46
48
  /**
@@ -148,7 +150,9 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
148
150
  return { models: cachedModels, stale: false };
149
151
  }
150
152
  const repairedModels = mergeDynamicModels(staticModels, cachedModels);
151
- writeModelCache(options.providerId, now(), repairedModels, true, staticFingerprint, dbPath);
153
+ if (options.canPublishCache?.() ?? true) {
154
+ writeModelCache(options.providerId, now(), repairedModels, true, staticFingerprint, dbPath);
155
+ }
152
156
  return { models: repairedModels, stale: false };
153
157
  }
154
158
 
@@ -169,24 +173,28 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
169
173
  const snapshotModels = applyFinalCodexGpt56ContextCap(
170
174
  mergeDynamicModels(mergeModelSources(staticModels, modelsDevModels), dynamicModels),
171
175
  );
172
- writeModelCache(options.providerId, now(), snapshotModels, true, staticFingerprint, dbPath);
176
+ if (options.canPublishCache?.() ?? true) {
177
+ writeModelCache(options.providerId, now(), snapshotModels, true, staticFingerprint, dbPath);
178
+ }
173
179
  } else {
174
180
  // Dynamic fetch failed — update cache with a non-authoritative snapshot so
175
181
  // stale state remains visible while retry backoff still applies.
176
182
  const latestCache = readModelCache<TApi>(options.providerId, ttlMs, now, dbPath);
177
- writeModelCache(
178
- options.providerId,
179
- now(),
180
- applyFinalCodexGpt56ContextCap(
181
- mergeDynamicModels(
182
- mergeModelSources(staticModels, modelsDevModels),
183
- normalizeModelList<TApi>(latestCache?.models ?? cache?.models ?? []),
183
+ if (options.canPublishCache?.() ?? true) {
184
+ writeModelCache(
185
+ options.providerId,
186
+ now(),
187
+ applyFinalCodexGpt56ContextCap(
188
+ mergeDynamicModels(
189
+ mergeModelSources(staticModels, modelsDevModels),
190
+ normalizeModelList<TApi>(latestCache?.models ?? cache?.models ?? []),
191
+ ),
184
192
  ),
185
- ),
186
- false,
187
- staticFingerprint,
188
- dbPath,
189
- );
193
+ false,
194
+ staticFingerprint,
195
+ dbPath,
196
+ );
197
+ }
190
198
  }
191
199
  }
192
200
  return {