@gajae-code/ai 0.11.1 → 0.11.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/dist/types/auth-storage.d.ts +2 -0
- package/dist/types/model-manager.d.ts +2 -0
- package/dist/types/providers/anthropic.d.ts +2 -1
- package/dist/types/providers/pi-native-server.d.ts +9 -17
- package/dist/types/types.d.ts +33 -0
- package/dist/types/utils/fallback-transport.d.ts +13 -2
- package/dist/types/utils/overflow.d.ts +4 -54
- package/package.json +2 -2
- package/src/auth-storage.ts +14 -3
- package/src/model-manager.ts +22 -14
- package/src/providers/anthropic-messages-server.ts +178 -17
- package/src/providers/anthropic.ts +193 -213
- package/src/providers/openai-chat-server.ts +93 -5
- package/src/providers/openai-codex-responses.ts +76 -13
- package/src/providers/openai-responses-server.ts +111 -36
- package/src/providers/openai-responses-shared.ts +71 -12
- package/src/providers/pi-native-server.ts +273 -20
- package/src/types.ts +19 -0
- package/src/utils/event-stream.ts +9 -5
- package/src/utils/fallback-transport.ts +111 -27
- package/src/utils/overflow.ts +72 -31
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.11.2] - 2026-07-19
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- `transportFailureFacts` now reduces transport headers to a plain record containing only the retained retry signals (`retry-after`, `retry-after-ms`). Providers attach these facts to error `AssistantMessage`s, and the previous shape carried the live fetch/SDK `Headers` instance — which is not structured-cloneable (`structuredClone` throws `DataCloneError`, "The object can not be cloned." under Bun) and not JSON-serializable (persisted as `{}` in session files, silently dropping the retry hint). Under a managed model fallback chain, snapshotting such an error message replaced the real provider failure with the local clone error and exhausted the whole chain. Normalization is idempotent (re-running facts on facts is structurally stable; errors carrying only unretained headers with no status/code now yield no facts instead of an empty facts object), Retry-After classification (`classifyFallbackTrigger`) is unchanged, and arbitrary response headers no longer reach persisted facts.
|
|
10
|
+
|
|
5
11
|
## [0.11.0] - 2026-07-15
|
|
6
12
|
### Added
|
|
7
13
|
|
|
@@ -410,6 +410,8 @@ export declare class AuthStorage {
|
|
|
410
410
|
* Remove a runtime API key override.
|
|
411
411
|
*/
|
|
412
412
|
removeRuntimeApiKey(provider: string): void;
|
|
413
|
+
/** Whether a provider is currently authenticated by a runtime API-key override. */
|
|
414
|
+
hasRuntimeApiKey(provider: string): boolean;
|
|
413
415
|
/**
|
|
414
416
|
* Register a per-provider API key sourced from user configuration
|
|
415
417
|
* (e.g. `models.yml` `providers.<name>.apiKey`). Higher priority than
|
|
@@ -30,6 +30,8 @@ export interface ModelManagerOptions<TApi extends Api = Api, TModelsDevPayload =
|
|
|
30
30
|
modelsDev?: ModelsDevFallback<TApi, TModelsDevPayload>;
|
|
31
31
|
/** Clock override for deterministic tests. */
|
|
32
32
|
now?: () => number;
|
|
33
|
+
/** Optional guard that must permit cache publication. Default: writes are permitted. */
|
|
34
|
+
canPublishCache?: () => boolean;
|
|
33
35
|
}
|
|
34
36
|
/**
|
|
35
37
|
* Resolution result.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import Anthropic, { type ClientOptions as AnthropicSdkClientOptions } from "@anthropic-ai/sdk";
|
|
2
|
-
import type { MessageParam } from "@anthropic-ai/sdk/resources/messages";
|
|
2
|
+
import type { MessageCreateParamsStreaming, MessageParam } from "@anthropic-ai/sdk/resources/messages";
|
|
3
3
|
import type { FetchImpl, Message, Model, ProviderSessionState, ServiceTier, SimpleStreamOptions, StreamFunction, StreamOptions, Usage } from "../types";
|
|
4
4
|
export type AnthropicHeaderOptions = {
|
|
5
5
|
apiKey: string;
|
|
@@ -181,6 +181,7 @@ type SystemBlockOptions = {
|
|
|
181
181
|
export declare function buildAnthropicSystemBlocks(systemPrompt: readonly string[] | undefined, options?: SystemBlockOptions): AnthropicSystemBlock[] | undefined;
|
|
182
182
|
export declare function normalizeExtraBetas(betas?: string[] | string): string[];
|
|
183
183
|
export declare function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): AnthropicClientOptionsResult;
|
|
184
|
+
export declare function normalizeCacheControlTtlOrdering(params: MessageCreateParamsStreaming): void;
|
|
184
185
|
export declare function convertAnthropicMessages(messages: Message[], model: Model<"anthropic-messages">, isOAuthToken: boolean, options?: {
|
|
185
186
|
repairLatestAssistantThinking?: boolean;
|
|
186
187
|
}): MessageParam[];
|
|
@@ -10,13 +10,12 @@
|
|
|
10
10
|
* translations impose on first-class pi-ai fields (service tier, cache
|
|
11
11
|
* markers, thinking budgets, tool-choice variants, …).
|
|
12
12
|
*
|
|
13
|
-
* The streaming wire is {@link AssistantMessageEvent} serialized
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
* actual cost.
|
|
13
|
+
* The streaming wire is {@link AssistantMessageEvent} serialized as SSE. Public
|
|
14
|
+
* projections omit private raw reasoning and serialized Responses reasoning
|
|
15
|
+
* signatures while preserving provider-displayable summaries and genuine opaque
|
|
16
|
+
* signatures. Including `partial: AssistantMessage` on every delta is O(N²) in
|
|
17
|
+
* turn length on the wire — acceptable for the loopback / sidecar topology this
|
|
18
|
+
* transport is designed for; provider latency dominates the actual cost.
|
|
20
19
|
*
|
|
21
20
|
* Endpoint contract:
|
|
22
21
|
* POST /v1/pi/stream
|
|
@@ -45,16 +44,9 @@ export interface PiNativeParsedRequest {
|
|
|
45
44
|
*/
|
|
46
45
|
export declare function parseRequest(body: unknown, _headers?: Headers): PiNativeParsedRequest;
|
|
47
46
|
/**
|
|
48
|
-
* Ship
|
|
49
|
-
*
|
|
50
|
-
*
|
|
51
|
-
* canonical event type IS the wire type. Including the rolling
|
|
52
|
-
* `partial: AssistantMessage` on every delta is quadratic in turn length
|
|
53
|
-
* on the wire, but for the loopback / sidecar topology this transport
|
|
54
|
-
* targets (containerized GJC → host gateway) the bandwidth cost is negligible
|
|
55
|
-
* compared to provider latency —
|
|
56
|
-
* and the client gets to feed the events straight into its existing
|
|
57
|
-
* `AssistantMessageEventStream.push()` plumbing with zero translation.
|
|
47
|
+
* Ship only public-safe {@link AssistantMessageEvent} projections. Unknown
|
|
48
|
+
* thinking blocks remain buffered until their terminal partial establishes that
|
|
49
|
+
* the provider-native block is safe; raw and mixed blocks never reach SSE.
|
|
58
50
|
*/
|
|
59
51
|
export declare function encodeStream(events: AssistantMessageEventStream): ReadableStream<Uint8Array>;
|
|
60
52
|
/**
|
package/dist/types/types.d.ts
CHANGED
|
@@ -319,6 +319,9 @@ export interface ThinkingContent {
|
|
|
319
319
|
thinking: string;
|
|
320
320
|
thinkingSignature?: string;
|
|
321
321
|
itemId?: string;
|
|
322
|
+
readonly provenance?: "summary" | "raw" | "mixed";
|
|
323
|
+
readonly summaryText?: string;
|
|
324
|
+
readonly rawText?: string;
|
|
322
325
|
}
|
|
323
326
|
export interface RedactedThinkingContent {
|
|
324
327
|
type: "redactedThinking";
|
|
@@ -542,6 +545,16 @@ export interface Tool<TParameters extends TSchema = TSchema> {
|
|
|
542
545
|
* calls route correctly. Absent for regular JSON function tools.
|
|
543
546
|
*/
|
|
544
547
|
customWireName?: string;
|
|
548
|
+
/**
|
|
549
|
+
* Optional safe projection for tool arguments or results. Extensions use this
|
|
550
|
+
* only for explicitly opt-in, display-safe summaries.
|
|
551
|
+
*/
|
|
552
|
+
safeSummary?: (kind: "args" | "result", value: unknown) => string | undefined;
|
|
553
|
+
/** Allowlisted argument/result field names for a safe fallback summary. */
|
|
554
|
+
safeSummaryFields?: {
|
|
555
|
+
args?: string[];
|
|
556
|
+
result?: string[];
|
|
557
|
+
};
|
|
545
558
|
}
|
|
546
559
|
export interface Context {
|
|
547
560
|
systemPrompt?: string[];
|
|
@@ -580,6 +593,20 @@ export type AssistantMessageEvent = {
|
|
|
580
593
|
contentIndex: number;
|
|
581
594
|
content: string;
|
|
582
595
|
partial: AssistantMessage;
|
|
596
|
+
} | {
|
|
597
|
+
type: "reasoning_summary_start";
|
|
598
|
+
contentIndex: number;
|
|
599
|
+
partial: AssistantMessage;
|
|
600
|
+
} | {
|
|
601
|
+
type: "reasoning_summary_delta";
|
|
602
|
+
contentIndex: number;
|
|
603
|
+
delta: string;
|
|
604
|
+
partial: AssistantMessage;
|
|
605
|
+
} | {
|
|
606
|
+
type: "reasoning_summary_end";
|
|
607
|
+
contentIndex: number;
|
|
608
|
+
content: string;
|
|
609
|
+
partial: AssistantMessage;
|
|
583
610
|
} | {
|
|
584
611
|
type: "toolcall_start";
|
|
585
612
|
contentIndex: number;
|
|
@@ -731,6 +758,12 @@ export interface AnthropicCompat extends ToolChoiceCompat {
|
|
|
731
758
|
supportsForcedToolChoice?: boolean;
|
|
732
759
|
/** Whether long prompt-cache retention (`ttl: "1h"`) is supported. Default: true for canonical Anthropic API. */
|
|
733
760
|
supportsLongCacheRetention?: boolean;
|
|
761
|
+
/**
|
|
762
|
+
* Prompt-cache transport accepted by this Anthropic-compatible endpoint.
|
|
763
|
+
* Canonical Anthropic defaults to `"automatic"`; noncanonical endpoints default
|
|
764
|
+
* to `"none"` and must explicitly opt into generated `"explicit"` markers.
|
|
765
|
+
*/
|
|
766
|
+
promptCacheMode?: "none" | "explicit" | "automatic";
|
|
734
767
|
}
|
|
735
768
|
/**
|
|
736
769
|
* OpenRouter provider routing preferences.
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
export type FallbackTriggerClass = "rate_limit" | "quota" | "auth" | "server" | "other";
|
|
1
|
+
export type FallbackTriggerClass = "rate_limit" | "quota" | "auth" | "server" | "unknown" | "other";
|
|
2
2
|
export interface FallbackTrigger {
|
|
3
3
|
class: FallbackTriggerClass;
|
|
4
4
|
retryAfterMs?: number;
|
|
@@ -7,12 +7,23 @@ export type TransportHeaders = Headers | Record<string, string | undefined>;
|
|
|
7
7
|
/**
|
|
8
8
|
* Structured facts from an upstream HTTP or transport failure. Retry decisions
|
|
9
9
|
* must use these facts rather than provider- or application-owned error text.
|
|
10
|
+
*
|
|
11
|
+
* `headers` is always a plain record limited to the retained retry-signal
|
|
12
|
+
* entries: facts travel on persisted `AssistantMessage`s and through
|
|
13
|
+
* `structuredClone` snapshots (managed fallback attempt staging), so they must
|
|
14
|
+
* never carry a live `Headers` instance — cloning one throws `DataCloneError`
|
|
15
|
+
* ("The object can not be cloned.") and masks the real provider failure.
|
|
10
16
|
*/
|
|
11
17
|
export interface TransportFailureFacts {
|
|
12
18
|
kind: "transport";
|
|
13
19
|
status?: number;
|
|
20
|
+
/** Canonical provider error code used for fallback classification. */
|
|
14
21
|
providerCode?: string;
|
|
15
|
-
|
|
22
|
+
/** Anthropic's typed `error.type`, preserved separately at the transport boundary. */
|
|
23
|
+
anthropicErrorType?: string;
|
|
24
|
+
/** OpenAI's typed `error.code`, preserved separately at the transport boundary. */
|
|
25
|
+
openaiErrorCode?: string;
|
|
26
|
+
headers?: Record<string, string>;
|
|
16
27
|
}
|
|
17
28
|
/** Opaque per-invocation marker required by managed fallback transport calls. */
|
|
18
29
|
export interface FallbackAttemptToken {
|
|
@@ -1,61 +1,11 @@
|
|
|
1
1
|
import type { AssistantMessage } from "../types";
|
|
2
|
+
import type { TransportFailureFacts } from "./fallback-transport";
|
|
3
|
+
export declare function classifyContextOverflow(message: AssistantMessage, transportFailure?: TransportFailureFacts, contextWindow?: number): boolean;
|
|
2
4
|
/**
|
|
3
5
|
* Check if an assistant message represents a context overflow error.
|
|
4
6
|
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
* specific error message pattern.
|
|
8
|
-
* 2. Silent overflow: Some providers accept overflow requests and return
|
|
9
|
-
* successfully. For these, we check if usage.input exceeds the context window.
|
|
10
|
-
* 3. Proxy-level overflow: Some proxies (e.g. LiteLLM) return a "successful"
|
|
11
|
-
* response with empty content and a fabricated near-zero usage when the
|
|
12
|
-
* upstream model's context window is exceeded.
|
|
13
|
-
*
|
|
14
|
-
* ## Reliability by Provider
|
|
15
|
-
*
|
|
16
|
-
* **Reliable detection (returns error with detectable message):**
|
|
17
|
-
* - Anthropic: "prompt is too long: X tokens > Y maximum"
|
|
18
|
-
* - OpenAI (Completions & Responses): "exceeds the context window"
|
|
19
|
-
* - Google Gemini: "input token count exceeds the maximum"
|
|
20
|
-
* - xAI (Grok): "maximum prompt length is X but request contains Y"
|
|
21
|
-
* - Groq: "reduce the length of the messages"
|
|
22
|
-
* - Cerebras: 400/413 status code (no body)
|
|
23
|
-
* - Mistral: 400/413 status code (no body)
|
|
24
|
-
* - HTTP 413 payload/entity-too-large variants
|
|
25
|
-
* - OpenRouter (all backends): "maximum context length is X tokens"
|
|
26
|
-
* - llama.cpp: "exceeds the available context size"
|
|
27
|
-
* - LM Studio: "greater than the context length"
|
|
28
|
-
* - Kimi For Coding: "exceeded model token limit: X (requested: Y)"
|
|
29
|
-
* - Anthropic 413: "request_too_large" (request body exceeds size limit)
|
|
30
|
-
* - HTTP 413: "Payload Too Large" / "Request Entity Too Large"
|
|
31
|
-
*
|
|
32
|
-
* **Unreliable detection:**
|
|
33
|
-
* - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
|
|
34
|
-
* sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.
|
|
35
|
-
* - Ollama: Silently truncates input without error. Cannot be detected via this function.
|
|
36
|
-
* - LiteLLM proxy: Returns a "successful" response with empty content and a
|
|
37
|
-
* fabricated near-zero usage (e.g. input: 1, output: 1) when the upstream
|
|
38
|
-
* model's context window is exceeded. Detected via Case 3 (empty content +
|
|
39
|
-
* anomalously low usage). Note: the LiteLLM proxy's context limit may differ
|
|
40
|
-
* from the underlying model's advertised contextWindow (e.g. configured via
|
|
41
|
-
* `model_info.max_tokens` in LiteLLM's config.yaml), so Case 2 (which compares
|
|
42
|
-
* usage.input against contextWindow) may not catch it.
|
|
43
|
-
* The response will have usage.input < expected, but we don't know the expected value.
|
|
44
|
-
*
|
|
45
|
-
* ## Custom Providers
|
|
46
|
-
*
|
|
47
|
-
* If you've added custom models via settings.json, this function may not detect
|
|
48
|
-
* overflow errors from those providers. To add support:
|
|
49
|
-
*
|
|
50
|
-
* 1. Send a request that exceeds the model's context window
|
|
51
|
-
* 2. Check the errorMessage in the response
|
|
52
|
-
* 3. Create a regex pattern that matches the error
|
|
53
|
-
* 4. The pattern should be added to OVERFLOW_PATTERNS in this file, or
|
|
54
|
-
* check the errorMessage yourself before calling this function
|
|
55
|
-
*
|
|
56
|
-
* @param message - The assistant message to check
|
|
57
|
-
* @param contextWindow - Optional context window size for detecting silent overflow (z.ai)
|
|
58
|
-
* @returns true if the message indicates a context overflow
|
|
7
|
+
* Callers with normalized transport facts should use {@link classifyContextOverflow}
|
|
8
|
+
* so typed provider codes take precedence over error prose.
|
|
59
9
|
*/
|
|
60
10
|
export declare function isContextOverflow(message: AssistantMessage, contextWindow?: number): boolean;
|
|
61
11
|
/**
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.11.
|
|
4
|
+
"version": "0.11.3",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo and Gajae Code Contributors",
|
|
@@ -40,7 +40,7 @@
|
|
|
40
40
|
"dependencies": {
|
|
41
41
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
42
42
|
"@bufbuild/protobuf": "^2.12.0",
|
|
43
|
-
"@gajae-code/utils": "0.11.
|
|
43
|
+
"@gajae-code/utils": "0.11.3",
|
|
44
44
|
"openai": "^6.36.0",
|
|
45
45
|
"partial-json": "^0.1.7",
|
|
46
46
|
"zod": "4.4.3"
|
package/src/auth-storage.ts
CHANGED
|
@@ -775,7 +775,9 @@ export class AuthStorage {
|
|
|
775
775
|
*/
|
|
776
776
|
static async create(dbPath: string, options: AuthStorageOptions = {}): Promise<AuthStorage> {
|
|
777
777
|
const store = await SqliteAuthCredentialStore.open(dbPath);
|
|
778
|
-
|
|
778
|
+
const storage = new AuthStorage(store, options);
|
|
779
|
+
await storage.reload();
|
|
780
|
+
return storage;
|
|
779
781
|
}
|
|
780
782
|
|
|
781
783
|
/**
|
|
@@ -854,6 +856,7 @@ export class AuthStorage {
|
|
|
854
856
|
*/
|
|
855
857
|
setRuntimeApiKey(provider: string, apiKey: string): void {
|
|
856
858
|
this.#runtimeOverrides.set(provider, apiKey);
|
|
859
|
+
this.#bumpGeneration("set-runtime-api-key");
|
|
857
860
|
}
|
|
858
861
|
|
|
859
862
|
/**
|
|
@@ -877,7 +880,12 @@ export class AuthStorage {
|
|
|
877
880
|
* Remove a runtime API key override.
|
|
878
881
|
*/
|
|
879
882
|
removeRuntimeApiKey(provider: string): void {
|
|
880
|
-
this.#runtimeOverrides.delete(provider);
|
|
883
|
+
if (this.#runtimeOverrides.delete(provider)) this.#bumpGeneration("remove-runtime-api-key");
|
|
884
|
+
}
|
|
885
|
+
|
|
886
|
+
/** Whether a provider is currently authenticated by a runtime API-key override. */
|
|
887
|
+
hasRuntimeApiKey(provider: string): boolean {
|
|
888
|
+
return Boolean(this.#runtimeOverrides.get(provider));
|
|
881
889
|
}
|
|
882
890
|
|
|
883
891
|
/**
|
|
@@ -892,13 +900,14 @@ export class AuthStorage {
|
|
|
892
900
|
*/
|
|
893
901
|
setConfigApiKey(provider: string, apiKey: string): void {
|
|
894
902
|
this.#configOverrides.set(provider, apiKey);
|
|
903
|
+
this.#bumpGeneration("set-config-api-key");
|
|
895
904
|
}
|
|
896
905
|
|
|
897
906
|
/**
|
|
898
907
|
* Remove a single config-sourced API key override.
|
|
899
908
|
*/
|
|
900
909
|
removeConfigApiKey(provider: string): void {
|
|
901
|
-
this.#configOverrides.delete(provider);
|
|
910
|
+
if (this.#configOverrides.delete(provider)) this.#bumpGeneration("remove-config-api-key");
|
|
902
911
|
}
|
|
903
912
|
|
|
904
913
|
/**
|
|
@@ -906,7 +915,9 @@ export class AuthStorage {
|
|
|
906
915
|
* re-parsing `models.yml` so removed entries actually disappear.
|
|
907
916
|
*/
|
|
908
917
|
clearConfigApiKeys(): void {
|
|
918
|
+
if (this.#configOverrides.size === 0) return;
|
|
909
919
|
this.#configOverrides.clear();
|
|
920
|
+
this.#bumpGeneration("clear-config-api-keys");
|
|
910
921
|
}
|
|
911
922
|
|
|
912
923
|
/**
|
package/src/model-manager.ts
CHANGED
|
@@ -41,6 +41,8 @@ export interface ModelManagerOptions<TApi extends Api = Api, TModelsDevPayload =
|
|
|
41
41
|
modelsDev?: ModelsDevFallback<TApi, TModelsDevPayload>;
|
|
42
42
|
/** Clock override for deterministic tests. */
|
|
43
43
|
now?: () => number;
|
|
44
|
+
/** Optional guard that must permit cache publication. Default: writes are permitted. */
|
|
45
|
+
canPublishCache?: () => boolean;
|
|
44
46
|
}
|
|
45
47
|
|
|
46
48
|
/**
|
|
@@ -148,7 +150,9 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
|
|
|
148
150
|
return { models: cachedModels, stale: false };
|
|
149
151
|
}
|
|
150
152
|
const repairedModels = mergeDynamicModels(staticModels, cachedModels);
|
|
151
|
-
|
|
153
|
+
if (options.canPublishCache?.() ?? true) {
|
|
154
|
+
writeModelCache(options.providerId, now(), repairedModels, true, staticFingerprint, dbPath);
|
|
155
|
+
}
|
|
152
156
|
return { models: repairedModels, stale: false };
|
|
153
157
|
}
|
|
154
158
|
|
|
@@ -169,24 +173,28 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
|
|
|
169
173
|
const snapshotModels = applyFinalCodexGpt56ContextCap(
|
|
170
174
|
mergeDynamicModels(mergeModelSources(staticModels, modelsDevModels), dynamicModels),
|
|
171
175
|
);
|
|
172
|
-
|
|
176
|
+
if (options.canPublishCache?.() ?? true) {
|
|
177
|
+
writeModelCache(options.providerId, now(), snapshotModels, true, staticFingerprint, dbPath);
|
|
178
|
+
}
|
|
173
179
|
} else {
|
|
174
180
|
// Dynamic fetch failed — update cache with a non-authoritative snapshot so
|
|
175
181
|
// stale state remains visible while retry backoff still applies.
|
|
176
182
|
const latestCache = readModelCache<TApi>(options.providerId, ttlMs, now, dbPath);
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
183
|
+
if (options.canPublishCache?.() ?? true) {
|
|
184
|
+
writeModelCache(
|
|
185
|
+
options.providerId,
|
|
186
|
+
now(),
|
|
187
|
+
applyFinalCodexGpt56ContextCap(
|
|
188
|
+
mergeDynamicModels(
|
|
189
|
+
mergeModelSources(staticModels, modelsDevModels),
|
|
190
|
+
normalizeModelList<TApi>(latestCache?.models ?? cache?.models ?? []),
|
|
191
|
+
),
|
|
184
192
|
),
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
193
|
+
false,
|
|
194
|
+
staticFingerprint,
|
|
195
|
+
dbPath,
|
|
196
|
+
);
|
|
197
|
+
}
|
|
190
198
|
}
|
|
191
199
|
}
|
|
192
200
|
return {
|
|
@@ -417,6 +417,62 @@ function mapStopReasonOut(reason: StopReason): "end_turn" | "max_tokens" | "tool
|
|
|
417
417
|
}
|
|
418
418
|
}
|
|
419
419
|
|
|
420
|
+
/**
|
|
421
|
+
* True when `signature` is one of OUR serialized OpenAI Responses reasoning-item
|
|
422
|
+
* envelopes (produced by the Responses/Codex decoders as
|
|
423
|
+
* `JSON.stringify(reasoningItem)` with `type: "reasoning"`). Such an envelope is
|
|
424
|
+
* NOT a valid Anthropic thinking signature and may embed raw chain-of-thought in
|
|
425
|
+
* `content[]`, so it must never be forwarded on Anthropic egress. Genuine opaque
|
|
426
|
+
* Anthropic signatures and any non-envelope string return false (fail safe).
|
|
427
|
+
*/
|
|
428
|
+
function isSerializedResponsesReasoningItem(signature: string): boolean {
|
|
429
|
+
try {
|
|
430
|
+
const parsed: unknown = JSON.parse(signature);
|
|
431
|
+
return (
|
|
432
|
+
typeof parsed === "object" &&
|
|
433
|
+
parsed !== null &&
|
|
434
|
+
!Array.isArray(parsed) &&
|
|
435
|
+
(parsed as { type?: unknown }).type === "reasoning"
|
|
436
|
+
);
|
|
437
|
+
} catch {
|
|
438
|
+
return false;
|
|
439
|
+
}
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
/**
|
|
443
|
+
* A thinking-block signature safe to emit on Anthropic egress: cross-protocol
|
|
444
|
+
* serialized Responses reasoning envelopes are omitted (they can carry raw CoT);
|
|
445
|
+
* genuine opaque Anthropic signatures round-trip unchanged.
|
|
446
|
+
*/
|
|
447
|
+
function safeThinkingSignature(signature: string | undefined): string | undefined {
|
|
448
|
+
if (!signature) return undefined;
|
|
449
|
+
if (isSerializedResponsesReasoningItem(signature)) return undefined;
|
|
450
|
+
return signature;
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
function isResponsesFamilyApi(api: AssistantMessage["api"]): boolean {
|
|
454
|
+
return api === "openai-responses" || api === "openai-codex-responses";
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
function safeThinkingText(content: ThinkingContent, api: AssistantMessage["api"]): string | undefined {
|
|
458
|
+
if (isResponsesFamilyApi(api) && content.provenance === undefined) return undefined;
|
|
459
|
+
if (content.provenance === "raw") return undefined;
|
|
460
|
+
if (content.provenance === "mixed") return content.summaryText;
|
|
461
|
+
if (content.provenance === "summary") return content.summaryText ?? content.thinking;
|
|
462
|
+
return content.thinking;
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
function hasRawOrMixedThinking(partial: AssistantMessage, contentIndex: number): boolean {
|
|
466
|
+
const content = partial.content[contentIndex];
|
|
467
|
+
return content?.type === "thinking" && (content.provenance === "raw" || content.provenance === "mixed");
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
/** Responses-family reasoning is untrusted until output_item.done assigns provenance. */
|
|
471
|
+
function hasUnfinalizedResponsesThinking(partial: AssistantMessage, contentIndex: number): boolean {
|
|
472
|
+
const content = partial.content[contentIndex];
|
|
473
|
+
return content?.type !== "thinking" || (isResponsesFamilyApi(partial.api) && content.provenance === undefined);
|
|
474
|
+
}
|
|
475
|
+
|
|
420
476
|
function encodeContentBlocks(message: AssistantMessage): Record<string, unknown>[] {
|
|
421
477
|
const blocks: Record<string, unknown>[] = [];
|
|
422
478
|
for (const c of message.content) {
|
|
@@ -425,8 +481,11 @@ function encodeContentBlocks(message: AssistantMessage): Record<string, unknown>
|
|
|
425
481
|
blocks.push({ type: "text", text: c.text });
|
|
426
482
|
break;
|
|
427
483
|
case "thinking": {
|
|
428
|
-
const
|
|
429
|
-
if (
|
|
484
|
+
const thinking = safeThinkingText(c, message.api);
|
|
485
|
+
if (thinking === undefined) break;
|
|
486
|
+
const b: Record<string, unknown> = { type: "thinking", thinking };
|
|
487
|
+
const sig = safeThinkingSignature(c.thinkingSignature);
|
|
488
|
+
if (sig) b.signature = sig;
|
|
430
489
|
blocks.push(b);
|
|
431
490
|
break;
|
|
432
491
|
}
|
|
@@ -495,6 +554,12 @@ export function encodeStream(
|
|
|
495
554
|
const messageId = newMessageId();
|
|
496
555
|
let started = false;
|
|
497
556
|
const open = new Map<number, OpenBlock>();
|
|
557
|
+
// contentIndexes that already streamed a reasoning summary delta, so a
|
|
558
|
+
// final-only reasoning_summary_end does not duplicate streamed summary text.
|
|
559
|
+
const summaryDeltaSeen = new Set<number>();
|
|
560
|
+
// Responses assigns reasoning provenance only at output_item.done. Keep its
|
|
561
|
+
// pre-classification bytes out of this public compatibility stream.
|
|
562
|
+
const pendingThinkingDeltas = new Map<number, string[]>();
|
|
498
563
|
|
|
499
564
|
const ensureStart = (partial: AssistantMessage) => {
|
|
500
565
|
if (started) return;
|
|
@@ -524,6 +589,30 @@ export function encodeStream(
|
|
|
524
589
|
open.delete(index);
|
|
525
590
|
};
|
|
526
591
|
|
|
592
|
+
const openThinking = (partial: AssistantMessage, index: number) => {
|
|
593
|
+
if (open.has(index)) return;
|
|
594
|
+
ensureStart(partial);
|
|
595
|
+
open.set(index, { index, kind: "thinking" });
|
|
596
|
+
controller.enqueue(
|
|
597
|
+
sseFrame("content_block_start", {
|
|
598
|
+
type: "content_block_start",
|
|
599
|
+
index,
|
|
600
|
+
content_block: { type: "thinking", thinking: "" },
|
|
601
|
+
}),
|
|
602
|
+
);
|
|
603
|
+
};
|
|
604
|
+
|
|
605
|
+
const writeThinkingDelta = (index: number, thinking: string) => {
|
|
606
|
+
if (thinking.length === 0) return;
|
|
607
|
+
controller.enqueue(
|
|
608
|
+
sseFrame("content_block_delta", {
|
|
609
|
+
type: "content_block_delta",
|
|
610
|
+
index,
|
|
611
|
+
delta: { type: "thinking_delta", thinking },
|
|
612
|
+
}),
|
|
613
|
+
);
|
|
614
|
+
};
|
|
615
|
+
|
|
527
616
|
try {
|
|
528
617
|
for await (const ev of events) {
|
|
529
618
|
switch (ev.type) {
|
|
@@ -555,18 +644,73 @@ export function encodeStream(
|
|
|
555
644
|
closeBlock(ev.contentIndex);
|
|
556
645
|
break;
|
|
557
646
|
case "thinking_start": {
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
647
|
+
if (hasRawOrMixedThinking(ev.partial, ev.contentIndex)) break;
|
|
648
|
+
if (hasUnfinalizedResponsesThinking(ev.partial, ev.contentIndex)) {
|
|
649
|
+
pendingThinkingDeltas.set(ev.contentIndex, []);
|
|
650
|
+
break;
|
|
651
|
+
}
|
|
652
|
+
openThinking(ev.partial, ev.contentIndex);
|
|
653
|
+
break;
|
|
654
|
+
}
|
|
655
|
+
case "thinking_delta": {
|
|
656
|
+
if (hasRawOrMixedThinking(ev.partial, ev.contentIndex)) break;
|
|
657
|
+
if (hasUnfinalizedResponsesThinking(ev.partial, ev.contentIndex)) {
|
|
658
|
+
const deltas = pendingThinkingDeltas.get(ev.contentIndex) ?? [];
|
|
659
|
+
deltas.push(ev.delta);
|
|
660
|
+
pendingThinkingDeltas.set(ev.contentIndex, deltas);
|
|
661
|
+
break;
|
|
662
|
+
}
|
|
663
|
+
writeThinkingDelta(ev.contentIndex, ev.delta);
|
|
664
|
+
break;
|
|
665
|
+
}
|
|
666
|
+
case "thinking_end": {
|
|
667
|
+
const unfinalized = hasUnfinalizedResponsesThinking(ev.partial, ev.contentIndex);
|
|
668
|
+
const rawOrMixed = hasRawOrMixedThinking(ev.partial, ev.contentIndex);
|
|
669
|
+
const pending = pendingThinkingDeltas.get(ev.contentIndex);
|
|
670
|
+
pendingThinkingDeltas.delete(ev.contentIndex);
|
|
671
|
+
if (rawOrMixed || unfinalized) {
|
|
672
|
+
closeBlock(ev.contentIndex);
|
|
673
|
+
break;
|
|
674
|
+
}
|
|
675
|
+
if (pending) {
|
|
676
|
+
openThinking(ev.partial, ev.contentIndex);
|
|
677
|
+
for (const delta of pending) writeThinkingDelta(ev.contentIndex, delta);
|
|
678
|
+
}
|
|
679
|
+
const c = ev.partial.content[ev.contentIndex];
|
|
680
|
+
const sig = c?.type === "thinking" ? safeThinkingSignature(c.thinkingSignature) : undefined;
|
|
681
|
+
if (sig) {
|
|
682
|
+
controller.enqueue(
|
|
683
|
+
sseFrame("content_block_delta", {
|
|
684
|
+
type: "content_block_delta",
|
|
685
|
+
index: ev.contentIndex,
|
|
686
|
+
delta: { type: "signature_delta", signature: sig },
|
|
687
|
+
}),
|
|
688
|
+
);
|
|
689
|
+
}
|
|
690
|
+
closeBlock(ev.contentIndex);
|
|
691
|
+
break;
|
|
692
|
+
}
|
|
693
|
+
case "reasoning_summary_start": {
|
|
694
|
+
// Provider-displayable summary reasoning surfaces as this format's
|
|
695
|
+
// native thinking channel. Open the thinking block if a thinking_start
|
|
696
|
+
// did not already (summary-only streams emit no thinking_start).
|
|
697
|
+
if (!open.has(ev.contentIndex)) {
|
|
698
|
+
ensureStart(ev.partial);
|
|
699
|
+
open.set(ev.contentIndex, { index: ev.contentIndex, kind: "thinking" });
|
|
700
|
+
controller.enqueue(
|
|
701
|
+
sseFrame("content_block_start", {
|
|
702
|
+
type: "content_block_start",
|
|
703
|
+
index: ev.contentIndex,
|
|
704
|
+
content_block: { type: "thinking", thinking: "" },
|
|
705
|
+
}),
|
|
706
|
+
);
|
|
707
|
+
}
|
|
567
708
|
break;
|
|
568
709
|
}
|
|
569
|
-
case "
|
|
710
|
+
case "reasoning_summary_delta":
|
|
711
|
+
// Only a non-whitespace delta counts as a delivered summary; a bare
|
|
712
|
+
// separator ("\n\n") must not suppress a later final-only end content.
|
|
713
|
+
if (ev.delta.trim().length > 0) summaryDeltaSeen.add(ev.contentIndex);
|
|
570
714
|
controller.enqueue(
|
|
571
715
|
sseFrame("content_block_delta", {
|
|
572
716
|
type: "content_block_delta",
|
|
@@ -575,18 +719,31 @@ export function encodeStream(
|
|
|
575
719
|
}),
|
|
576
720
|
);
|
|
577
721
|
break;
|
|
578
|
-
case "
|
|
579
|
-
|
|
580
|
-
|
|
722
|
+
case "reasoning_summary_end": {
|
|
723
|
+
// Final-only summary: text arrives only on the end event with no prior
|
|
724
|
+
// deltas. Ensure the thinking block is open, then surface the content as
|
|
725
|
+
// a thinking_delta (skip when deltas already streamed to avoid dup). The
|
|
726
|
+
// thinking block is closed by the subsequent thinking_end.
|
|
727
|
+
if (ev.content.length > 0 && !summaryDeltaSeen.has(ev.contentIndex)) {
|
|
728
|
+
if (!open.has(ev.contentIndex)) {
|
|
729
|
+
ensureStart(ev.partial);
|
|
730
|
+
open.set(ev.contentIndex, { index: ev.contentIndex, kind: "thinking" });
|
|
731
|
+
controller.enqueue(
|
|
732
|
+
sseFrame("content_block_start", {
|
|
733
|
+
type: "content_block_start",
|
|
734
|
+
index: ev.contentIndex,
|
|
735
|
+
content_block: { type: "thinking", thinking: "" },
|
|
736
|
+
}),
|
|
737
|
+
);
|
|
738
|
+
}
|
|
581
739
|
controller.enqueue(
|
|
582
740
|
sseFrame("content_block_delta", {
|
|
583
741
|
type: "content_block_delta",
|
|
584
742
|
index: ev.contentIndex,
|
|
585
|
-
delta: { type: "
|
|
743
|
+
delta: { type: "thinking_delta", thinking: ev.content },
|
|
586
744
|
}),
|
|
587
745
|
);
|
|
588
746
|
}
|
|
589
|
-
closeBlock(ev.contentIndex);
|
|
590
747
|
break;
|
|
591
748
|
}
|
|
592
749
|
case "toolcall_start": {
|
|
@@ -621,6 +778,7 @@ export function encodeStream(
|
|
|
621
778
|
break;
|
|
622
779
|
case "done": {
|
|
623
780
|
for (const idx of [...open.keys()]) closeBlock(idx);
|
|
781
|
+
pendingThinkingDeltas.clear();
|
|
624
782
|
controller.enqueue(
|
|
625
783
|
sseFrame("message_delta", {
|
|
626
784
|
type: "message_delta",
|
|
@@ -636,6 +794,7 @@ export function encodeStream(
|
|
|
636
794
|
}
|
|
637
795
|
case "error": {
|
|
638
796
|
const msg = ev.error.errorMessage ?? "stream error";
|
|
797
|
+
pendingThinkingDeltas.clear();
|
|
639
798
|
controller.enqueue(
|
|
640
799
|
sseFrame("error", { type: "error", error: { type: "api_error", message: msg } }),
|
|
641
800
|
);
|
|
@@ -645,10 +804,12 @@ export function encodeStream(
|
|
|
645
804
|
}
|
|
646
805
|
}
|
|
647
806
|
// stream ended without explicit done; close gracefully
|
|
807
|
+
pendingThinkingDeltas.clear();
|
|
648
808
|
for (const idx of [...open.keys()]) closeBlock(idx);
|
|
649
809
|
controller.enqueue(sseFrame("message_stop", { type: "message_stop" }));
|
|
650
810
|
controller.close();
|
|
651
811
|
} catch (err) {
|
|
812
|
+
pendingThinkingDeltas.clear();
|
|
652
813
|
controller.enqueue(
|
|
653
814
|
sseFrame("error", {
|
|
654
815
|
type: "error",
|