@caeliq/llms 1.0.69 → 1.0.71

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ export {};
@@ -48,7 +48,7 @@ export declare function modelIdForRequestedOneMillionBeta(modelId: string | unde
48
48
  * post-build Anthropic body (not the Unified request) because
49
49
  * `buildAnthropicBody` may synthesize `thinking`/`output_config` itself.
50
50
  */
51
- export declare function applyClaudeModelCapabilityAdjustments(anthropicBody: Record<string, any>, entry: ClaudeModelCatalogEntry | undefined): void;
51
+ export declare function applyClaudeModelCapabilityAdjustments(anthropicBody: Record<string, any>, entry: ClaudeModelCatalogEntry | undefined, logger?: any): void;
52
52
  /** Test-only reset hook so session-id state doesn't leak across test cases. */
53
53
  export declare function __resetClaudeAuthTransformerStateForTests(): void;
54
54
  /** Synthesized Claude Code identity headers for the non-Claude-Code branch. */
@@ -5,23 +5,37 @@ export declare class OpencodeHeadersTransformer implements Transformer {
5
5
  requestPhase: "transport";
6
6
  private lastTimestamp;
7
7
  private counter;
8
+ private firstEventTimeoutMs;
9
+ private firstProgressTimeoutMs;
10
+ private streamIdleTimeoutMs;
11
+ private reasoningIdleTimeoutMs;
8
12
  transformRequestIn(request: any, provider: any, context: any): Promise<Record<string, any>>;
9
- transformResponseOut(response: Response): Promise<Response>;
13
+ transformResponseOut(response: Response, context?: any): Promise<Response>;
14
+ /** Restore JSON for clients whose requests were streamed only to pass Zen's gate. */
15
+ private collectForcedStream;
10
16
  /**
11
- * Own the full upstream call so a `No provider available` 401 can be retried
12
- * with a fresh session in isolation. Any other non-ok response is re-thrown in
13
- * the exact shape sendRequestToProvider would have produced, so genuine
14
- * auth/rate/server errors keep flowing through the normal error + fallback
15
- * path unchanged.
17
+ * Own the full upstream call so Zen routing failures can be recovered by
18
+ * re-rolling the session, and transient failures retried on the same one.
19
+ * Exhausted routing failures become 503 so the normal fallback path can try
20
+ * another model; ordinary 4xx errors retain their upstream status.
16
21
  */
17
22
  private sendWithSessionRetry;
23
+ /**
24
+ * Zen can send SSE headers and lifecycle events, then stall before any
25
+ * output. Hold the response until a complete event and real output progress
26
+ * arrive while a fallback is still possible, then bound later idle reads.
27
+ * An open reasoning item emits nothing until it finishes when no summary is
28
+ * streamed, so while one is open the looser reasoning bound applies instead.
29
+ */
30
+ private requireFirstZenEvent;
18
31
  private preserveZenStreamErrors;
19
32
  private static zenStreamFailure;
20
33
  private buildHeaders;
21
34
  private resolveParentSessionId;
22
35
  /**
23
- * True only for the two Zen session-hash routing failures — a re-roll can
24
- * recover these. Kept deliberately narrow (exact status + message) so genuine
36
+ * True only for the two Zen session-hash routing failures — retried on the
37
+ * SAME session (which must not change during a conversation), then passed
38
+ * to fallback. Kept deliberately narrow (exact status + message) so genuine
25
39
  * auth errors (401 invalid key) and request errors (400 validation) are never
26
40
  * mistaken for routing failures and pass straight through.
27
41
  */
@@ -32,8 +46,77 @@ export declare class OpencodeHeadersTransformer implements Transformer {
32
46
  private retryAfterHeaders;
33
47
  private ensurePromptCacheKey;
34
48
  private ensurePromptCacheRetention;
49
+ /**
50
+ * Free-tier gate scope: `-free` models on a zen (opencode.ai) endpoint.
51
+ * Paid Zen models and non-Zen providers sharing this transformer skip the
52
+ * stub/stream handling below entirely.
53
+ */
54
+ private static isFreeTierZenRequest;
55
+ /**
56
+ * The free-tier gate answers `stream: false` with 403 FreeTierError even
57
+ * when everything else is exact (curl A/B 2026-09-24). Force SSE on the
58
+ * wire; the normal response path does not de-stream SSE for JSON clients,
59
+ * so transformRequestIn records the forced shape and transformResponseOut
60
+ * restores JSON via collectForcedStream.
61
+ */
62
+ private ensureStreamedForFreeTier;
63
+ /**
64
+ * Free-tier thinking arrives almost entirely as opaque `encrypted_content`;
65
+ * the only readable part is the reasoning summary, and Zen emits ~nothing
66
+ * unless `reasoning.summary` is asked for (curl A/B 2026-09-24, same prompt:
67
+ * `"detailed"` → 68 summary chars, `"auto"` → 0). Stamp `detailed` when the
68
+ * client already reasons but states no summary preference. An explicit
69
+ * client value (including `"none"`) always wins; absent/disabled reasoning
70
+ * is left alone so non-reasoning calls never gain a reasoning block.
71
+ */
72
+ private ensureDetailedSummaryForFreeTier;
73
+ /**
74
+ * Zen aborts free-tier Responses runs (`response.incomplete`, no further
75
+ * events — a client-side stall) unless `prompt_cache_key` equals the
76
+ * `x-opencode-session` header (curl A/B 2026-09-24). CCR's generic cache
77
+ * key (`ccr_<sha256>`, stable per conversation but foreign to Zen) must be
78
+ * replaced after `ensurePromptCacheKey` runs. Scoped to free-tier Zen on the
79
+ * Responses wire; chat bodies and paid models keep existing behavior.
80
+ * Runs per attempt so header and key stay in lockstep.
81
+ */
82
+ private applyFreeTierCacheKey;
83
+ /**
84
+ * Inject the exact-name `read`/`shell` function stubs the free-tier gate
85
+ * requires (curl A/B 2026-09-24: lowercase exact match; `Read`/`Bash`
86
+ * fail, schemas are free-form, extras harmless). Client tools are never
87
+ * modified or reordered; stubs are appended. Each stub clones its client
88
+ * counterpart's description/parameters when present so a
89
+ * stub call maps back onto a schema the client already accepts; otherwise
90
+ * it carries a minimal empty-object schema. Returns the alias map
91
+ * (stub -> client name, or null without counterpart) for the response
92
+ * stage; no aliases when nothing was injected.
93
+ */
94
+ private ensureGateStubTools;
95
+ private static readonly GATE_STUB_COUNTERPARTS;
96
+ private static outgoingToolShape;
97
+ private static toolName;
98
+ private static toolDef;
99
+ private static buildStubTool;
100
+ private static gateAliasesFrom;
101
+ /**
102
+ * Map gate-stub calls back to the advertised client tool inside SSE events.
103
+ * Only `name` fields on
104
+ * Responses `function_call` objects and chat `tool_calls[].function` objects
105
+ * are touched; text payloads and unrelated events pass through byte-identical.
106
+ * Stubs without a client counterpart are left alone (the client errors on
107
+ * them exactly as it would on any unknown tool).
108
+ */
109
+ private rewriteGateStubCalls;
110
+ private static renameGateStubCalls;
35
111
  private fingerprintConversation;
36
- private getOrCreateSessionId;
37
112
  private invalidateSession;
113
+ /**
114
+ * Conversation identity for the Zen session binding: an explicit client
115
+ * session id wherever the client supplies one (router-parsed, protocol
116
+ * context, or the shared header/body extractor used for cache keys), and
117
+ * the content fingerprint only for fully anonymous clients.
118
+ */
119
+ private resolveConversationId;
120
+ private getOrCreateSessionId;
38
121
  private generateId;
39
122
  }
@@ -37,6 +37,18 @@ export declare function inspectAnthropicClientFingerprint(headers: Record<string
37
37
  */
38
38
  export declare function getAnthropicProviderMode(provider: any, endpointTransformerName?: string): AnthropicProviderMode;
39
39
  export declare function isNativeAnthropicClient(kind: AnthropicClientKind): boolean;
40
+ /** `CLAUDE_AUTH_NATIVE_CACHE_TTL` values: `1h` (default) or `client`. */
41
+ export type NativeClaudeOAuthCacheTtlMode = "1h" | "client";
42
+ export declare function resolveNativeClaudeOAuthCacheTtlMode(value: unknown): NativeClaudeOAuthCacheTtlMode;
43
+ /**
44
+ * Native Desktop/CLI on claude-auth OAuth: extend the client's default-TTL
45
+ * (5m) ephemeral breakpoints to 1h, the TTL the third-party profile already
46
+ * uses on this route. Breakpoint placement is untouched. A body that already
47
+ * sets a TTL on any breakpoint is left exactly as sent (an explicit client
48
+ * choice), as is every body when the mode is `client`. Returns whether the
49
+ * body changed.
50
+ */
51
+ export declare function applyNativeClaudeOAuthCacheTtl(body: any, context: AnthropicClientPolicyContext | undefined, mode?: NativeClaudeOAuthCacheTtlMode): boolean;
40
52
  /**
41
53
  * Apply the one and only system transformation allowed by the gateway policy.
42
54
  * This runs after routing has identified an in-scope Anthropic destination and
@@ -5,10 +5,14 @@
5
5
  * transformer chain without a bisect.
6
6
  */
7
7
  export type CachePrefixStage = "client" | "wire";
8
+ /**
9
+ * Routing headers that pin prompt-cache affinity. `x-client-request-id` is
10
+ * deliberately absent: claude-auth mints a fresh one per request (as Claude
11
+ * Code does), and Codex sets it equal to `thread-id`, which is tracked here.
12
+ */
8
13
  export type CacheAffinityHeaders = {
9
14
  sessionId?: string;
10
15
  threadId?: string;
11
- clientRequestId?: string;
12
16
  };
13
17
  export type CachePrefixSegment = {
14
18
  path: string;
@@ -9,7 +9,7 @@ export declare function extractFirstUserMessageText(messages: UnifiedMessage[] |
9
9
  export declare function computeVersionSuffix(text: string, version: string): string;
10
10
  /**
11
11
  * Current first-party Anthropic requests use the literal cch marker. Older
12
- * captures showed a random session value; the current 2.1.226 decompilation
12
+ * captures showed a random session value; the current 2.1.280 decompilation
13
13
  * gates this field to first-party/Vertex and emits `00000`.
14
14
  */
15
15
  export declare function sessionCch(): string;
@@ -25,6 +25,15 @@ export declare function buildClaudeBillingHeaderValue(messages: UnifiedMessage[]
25
25
  export declare function normalizeSystemToArray(request: UnifiedChatRequest): TextContent[];
26
26
  /** Drop any existing billing entry (dedupe) and prepend a fresh one at system[0]. */
27
27
  export declare function applyClaudeBillingSystemBlock(system: TextContent[], messages: UnifiedMessage[] | undefined): void;
28
+ /**
29
+ * Drop any existing billing entry (dedupe) and reserve system[0] for a fresh
30
+ * one, so applyClaudeSystemIdentity places identity right after it. The value
31
+ * samples the first user text as sent, which relocation may still change, so
32
+ * fill it afterwards with fillClaudeBillingSystemBlock.
33
+ */
34
+ export declare function reserveClaudeBillingSystemBlock(system: TextContent[]): TextContent;
35
+ /** Fill a reserved billing block; remove it when attribution is disabled. */
36
+ export declare function fillClaudeBillingSystemBlock(system: TextContent[], block: TextContent, messages: UnifiedMessage[] | undefined): void;
28
37
  /**
29
38
  * Insert SYSTEM_IDENTITY after the billing block (normally system[1], or
30
39
  * system[0] when attribution is disabled). Any remaining caller system
@@ -45,8 +54,13 @@ export declare function applyClaudeSystemIdentity(system: TextContent[]): void;
45
54
  * A no-op when there is no user message to attach the content to, so nothing
46
55
  * is silently dropped — the caller's system content stays in `system[]`
47
56
  * instead.
57
+ *
58
+ * Returns the inserted block unless the first user message has non-empty
59
+ * string content (the relocated text is then a standalone block a cache
60
+ * breakpoint can end on); non-empty string content is prefixed in place and
61
+ * returns undefined.
48
62
  */
49
- export declare function relocateForeignSystemContent(system: TextContent[], messages: UnifiedMessage[] | undefined): void;
63
+ export declare function relocateForeignSystemContent(system: TextContent[], messages: UnifiedMessage[] | undefined): TextContent | undefined;
50
64
  /**
51
65
  * Claude Code's OAuth validator expects tool names in the mcp_PascalCase
52
66
  * spelling used by the official CLI. Non-Claude-Code clients commonly send
@@ -1,6 +1,6 @@
1
1
  /**
2
2
  * Claude Code's bundled model capability catalog, revalidated against the
3
- * installed v2.1.226 CLI decompilation.
3
+ * installed v2.1.280 CLI decompilation.
4
4
  *
5
5
  * Every model-dependent decision in the claude-auth impersonation path (beta
6
6
  * flags, effort support, thinking shape, max_tokens ceiling) is driven by
@@ -21,6 +21,21 @@ export interface ClaudeModelCatalogEntry {
21
21
  };
22
22
  defaultEffort?: string;
23
23
  capabilities: string[];
24
+ /**
25
+ * CCR-owned Messages API request constraints. Not part of Claude Code's
26
+ * catalog (Claude Code simply never sends these shapes); sourced from
27
+ * Anthropic's per-model API documentation. Each flag names a request shape
28
+ * the model rejects with HTTP 400, so third-party emulation must normalize
29
+ * it away before sending.
30
+ */
31
+ apiConstraints?: {
32
+ /** `thinking: {type: "disabled"}` is rejected; thinking is always on. */
33
+ thinkingAlwaysOn?: boolean;
34
+ /** Forced `tool_choice` (`any` / `tool`) is rejected. */
35
+ noForcedToolChoice?: boolean;
36
+ /** `temperature` / `top_p` / `top_k` are rejected. */
37
+ noSamplingParams?: boolean;
38
+ };
24
39
  }
25
40
  export declare const CLAUDE_MODEL_CATALOG: Record<string, ClaudeModelCatalogEntry>;
26
41
  /** Strip the "[1m]" wire marker, reporting whether it was present. */
@@ -37,8 +37,13 @@ export declare function toAnthropicReasoningEffort(effortValue: unknown): Exclud
37
37
  * `codex,gpt-6-astra`). Anchored so `gpt-60` / `gpt-5.6` do not match.
38
38
  */
39
39
  export declare function isGpt6FamilyModel(model: unknown): boolean;
40
- /** Astra rejects `none` / `minimal`; OpenAI's migration floor is `low`. */
40
+ /** Astra/Sol reject `none` / `minimal`; OpenAI's migration floor is `low`. */
41
41
  export declare function coerceGpt6ReasoningEffort(model: unknown, effort: ThinkLevel | undefined): ThinkLevel | undefined;
42
+ /**
43
+ * GPT-6 Luna slugs (`gpt-6-luna`, `openai/gpt-6-luna`, `codex,gpt-6-luna`).
44
+ * Anchored so `gpt-6-sol` / `gpt-6-astra` do not match.
45
+ */
46
+ export declare function isGpt6LunaModel(model: unknown): boolean;
42
47
  /**
43
48
  * Remap unsupported GPT-6 efforts on a Responses/Unified request in place.
44
49
  * Covers convert (`openai-responses`) and same-protocol wire-keep (`codex`).
@@ -39,6 +39,12 @@ export declare function unifiedToolTextOnly(content: unknown): string;
39
39
  export declare function unifiedToolHasMedia(content: unknown): boolean;
40
40
  /** Anthropic tool_result.content: string or (text|image|document)[]. */
41
41
  export declare function unifiedToolContentToAnthropic(content: unknown): string | any[];
42
+ /**
43
+ * Anthropic blocks for one Unified user content part. Empty when the part is
44
+ * not sent (empty text, image without URL, file without data or URL, unknown
45
+ * type), so callers placing cache breakpoints can skip it.
46
+ */
47
+ export declare function unifiedUserPartToAnthropic(part: any): any[];
42
48
  /**
43
49
  * Anthropic inbound tool_result.content → Unified tool content
44
50
  * (string or text/image_url/file parts).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@caeliq/llms",
3
- "version": "1.0.69",
3
+ "version": "1.0.71",
4
4
  "description": "A universal LLM API transformation server",
5
5
  "main": "dist/cjs/server.cjs",
6
6
  "module": "dist/esm/server.mjs",
@@ -30,8 +30,8 @@
30
30
  ],
31
31
  "dependencies": {
32
32
  "@anthropic-ai/sdk": "^0.120.0",
33
- "@caeliq/ccr-shared": "^2.1.11",
34
- "@cursor/sdk": "^1.0.30",
33
+ "@caeliq/ccr-shared": "^2.1.13",
34
+ "@cursor/sdk": "^1.0.32",
35
35
  "@fastify/cors": "^11.3.0",
36
36
  "@fastify/rate-limit": "^11.2.0",
37
37
  "@google/genai": "^2.18.0",