@caeliq/llms 1.0.69 → 1.0.70

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ export {};
@@ -48,7 +48,7 @@ export declare function modelIdForRequestedOneMillionBeta(modelId: string | unde
48
48
  * post-build Anthropic body (not the Unified request) because
49
49
  * `buildAnthropicBody` may synthesize `thinking`/`output_config` itself.
50
50
  */
51
- export declare function applyClaudeModelCapabilityAdjustments(anthropicBody: Record<string, any>, entry: ClaudeModelCatalogEntry | undefined): void;
51
+ export declare function applyClaudeModelCapabilityAdjustments(anthropicBody: Record<string, any>, entry: ClaudeModelCatalogEntry | undefined, logger?: any): void;
52
52
  /** Test-only reset hook so session-id state doesn't leak across test cases. */
53
53
  export declare function __resetClaudeAuthTransformerStateForTests(): void;
54
54
  /** Synthesized Claude Code identity headers for the non-Claude-Code branch. */
@@ -5,23 +5,37 @@ export declare class OpencodeHeadersTransformer implements Transformer {
5
5
  requestPhase: "transport";
6
6
  private lastTimestamp;
7
7
  private counter;
8
+ private firstEventTimeoutMs;
9
+ private firstProgressTimeoutMs;
10
+ private streamIdleTimeoutMs;
11
+ private reasoningIdleTimeoutMs;
8
12
  transformRequestIn(request: any, provider: any, context: any): Promise<Record<string, any>>;
9
- transformResponseOut(response: Response): Promise<Response>;
13
+ transformResponseOut(response: Response, context?: any): Promise<Response>;
14
+ /** Restore JSON for clients whose requests were streamed only to pass Zen's gate. */
15
+ private collectForcedStream;
10
16
  /**
11
- * Own the full upstream call so a `No provider available` 401 can be retried
12
- * with a fresh session in isolation. Any other non-ok response is re-thrown in
13
- * the exact shape sendRequestToProvider would have produced, so genuine
14
- * auth/rate/server errors keep flowing through the normal error + fallback
15
- * path unchanged.
17
+ * Own the full upstream call so Zen routing failures can be recovered by
18
+ * re-rolling the session, and transient failures retried on the same one.
19
+ * Exhausted routing failures become 503 so the normal fallback path can try
20
+ * another model; ordinary 4xx errors retain their upstream status.
16
21
  */
17
22
  private sendWithSessionRetry;
23
+ /**
24
+ * Zen can send SSE headers and lifecycle events, then stall before any
25
+ * output. Hold the response until a complete event and real output progress
26
+ * arrive while a fallback is still possible, then bound later idle reads.
27
+ * An open reasoning item emits nothing until it finishes when no summary is
28
+ * streamed, so while one is open the looser reasoning bound applies instead.
29
+ */
30
+ private requireFirstZenEvent;
18
31
  private preserveZenStreamErrors;
19
32
  private static zenStreamFailure;
20
33
  private buildHeaders;
21
34
  private resolveParentSessionId;
22
35
  /**
23
- * True only for the two Zen session-hash routing failures — a re-roll can
24
- * recover these. Kept deliberately narrow (exact status + message) so genuine
36
+ * True only for the two Zen session-hash routing failures — retried on the
37
+ * SAME session (which must not change during a conversation), then passed
38
+ * to fallback. Kept deliberately narrow (exact status + message) so genuine
25
39
  * auth errors (401 invalid key) and request errors (400 validation) are never
26
40
  * mistaken for routing failures and pass straight through.
27
41
  */
@@ -32,8 +46,77 @@ export declare class OpencodeHeadersTransformer implements Transformer {
32
46
  private retryAfterHeaders;
33
47
  private ensurePromptCacheKey;
34
48
  private ensurePromptCacheRetention;
49
+ /**
50
+ * Free-tier gate scope: `-free` models on a zen (opencode.ai) endpoint.
51
+ * Paid Zen models and non-Zen providers sharing this transformer skip the
52
+ * stub/stream handling below entirely.
53
+ */
54
+ private static isFreeTierZenRequest;
55
+ /**
56
+ * The free-tier gate answers `stream: false` with 403 FreeTierError even
57
+ * when everything else is exact (curl A/B 2026-09-24). Force SSE on the
58
+ * wire; the normal response path does not de-stream SSE for JSON clients,
59
+ * so transformRequestIn records the forced shape and transformResponseOut
60
+ * restores JSON via collectForcedStream.
61
+ */
62
+ private ensureStreamedForFreeTier;
63
+ /**
64
+ * Free-tier thinking arrives almost entirely as opaque `encrypted_content`;
65
+ * the only readable part is the reasoning summary, and Zen emits ~nothing
66
+ * unless `reasoning.summary` is asked for (curl A/B 2026-09-24, same prompt:
67
+ * `"detailed"` → 68 summary chars, `"auto"` → 0). Stamp `detailed` when the
68
+ * client already reasons but states no summary preference. An explicit
69
+ * client value (including `"none"`) always wins; absent/disabled reasoning
70
+ * is left alone so non-reasoning calls never gain a reasoning block.
71
+ */
72
+ private ensureDetailedSummaryForFreeTier;
73
+ /**
74
+ * Zen aborts free-tier Responses runs (`response.incomplete`, no further
75
+ * events — a client-side stall) unless `prompt_cache_key` equals the
76
+ * `x-opencode-session` header (curl A/B 2026-09-24). CCR's generic cache
77
+ * key (`ccr_<sha256>`, stable per conversation but foreign to Zen) must be
78
+ * replaced after `ensurePromptCacheKey` runs. Scoped to free-tier Zen on the
79
+ * Responses wire; chat bodies and paid models keep existing behavior.
80
+ * Runs per attempt so header and key stay in lockstep.
81
+ */
82
+ private applyFreeTierCacheKey;
83
+ /**
84
+ * Inject the exact-name `read`/`shell` function stubs the free-tier gate
85
+ * requires (curl A/B 2026-09-24: lowercase exact match; `Read`/`Bash`
86
+ * fail, schemas are free-form, extras harmless). Client tools are never
87
+ * modified or reordered; stubs are appended. Each stub clones its client
88
+ * counterpart's description/parameters when present so a
89
+ * stub call maps back onto a schema the client already accepts; otherwise
90
+ * it carries a minimal empty-object schema. Returns the alias map
91
+ * (stub -> client name, or null without counterpart) for the response
92
+ * stage; no aliases when nothing was injected.
93
+ */
94
+ private ensureGateStubTools;
95
+ private static readonly GATE_STUB_COUNTERPARTS;
96
+ private static outgoingToolShape;
97
+ private static toolName;
98
+ private static toolDef;
99
+ private static buildStubTool;
100
+ private static gateAliasesFrom;
101
+ /**
102
+ * Map gate-stub calls back to the advertised client tool inside SSE events.
103
+ * Only `name` fields on
104
+ * Responses `function_call` objects and chat `tool_calls[].function` objects
105
+ * are touched; text payloads and unrelated events pass through byte-identical.
106
+ * Stubs without a client counterpart are left alone (the client errors on
107
+ * them exactly as it would on any unknown tool).
108
+ */
109
+ private rewriteGateStubCalls;
110
+ private static renameGateStubCalls;
35
111
  private fingerprintConversation;
36
- private getOrCreateSessionId;
37
112
  private invalidateSession;
113
+ /**
114
+ * Conversation identity for the Zen session binding: an explicit client
115
+ * session id wherever the client supplies one (router-parsed, protocol
116
+ * context, or the shared header/body extractor used for cache keys), and
117
+ * the content fingerprint only for fully anonymous clients.
118
+ */
119
+ private resolveConversationId;
120
+ private getOrCreateSessionId;
38
121
  private generateId;
39
122
  }
@@ -9,7 +9,7 @@ export declare function extractFirstUserMessageText(messages: UnifiedMessage[] |
9
9
  export declare function computeVersionSuffix(text: string, version: string): string;
10
10
  /**
11
11
  * Current first-party Anthropic requests use the literal cch marker. Older
12
- * captures showed a random session value; the current 2.1.226 decompilation
12
+ * captures showed a random session value; the current 2.1.280 decompilation
13
13
  * gates this field to first-party/Vertex and emits `00000`.
14
14
  */
15
15
  export declare function sessionCch(): string;
@@ -1,6 +1,6 @@
1
1
  /**
2
2
  * Claude Code's bundled model capability catalog, revalidated against the
3
- * installed v2.1.226 CLI decompilation.
3
+ * installed v2.1.280 CLI decompilation.
4
4
  *
5
5
  * Every model-dependent decision in the claude-auth impersonation path (beta
6
6
  * flags, effort support, thinking shape, max_tokens ceiling) is driven by
@@ -21,6 +21,21 @@ export interface ClaudeModelCatalogEntry {
21
21
  };
22
22
  defaultEffort?: string;
23
23
  capabilities: string[];
24
+ /**
25
+ * CCR-owned Messages API request constraints. Not part of Claude Code's
26
+ * catalog (Claude Code simply never sends these shapes); sourced from
27
+ * Anthropic's per-model API documentation. Each flag names a request shape
28
+ * the model rejects with HTTP 400, so third-party emulation must normalize
29
+ * it away before sending.
30
+ */
31
+ apiConstraints?: {
32
+ /** `thinking: {type: "disabled"}` is rejected; thinking is always on. */
33
+ thinkingAlwaysOn?: boolean;
34
+ /** Forced `tool_choice` (`any` / `tool`) is rejected. */
35
+ noForcedToolChoice?: boolean;
36
+ /** `temperature` / `top_p` / `top_k` are rejected. */
37
+ noSamplingParams?: boolean;
38
+ };
24
39
  }
25
40
  export declare const CLAUDE_MODEL_CATALOG: Record<string, ClaudeModelCatalogEntry>;
26
41
  /** Strip the "[1m]" wire marker, reporting whether it was present. */
@@ -37,8 +37,13 @@ export declare function toAnthropicReasoningEffort(effortValue: unknown): Exclud
37
37
  * `codex,gpt-6-astra`). Anchored so `gpt-60` / `gpt-5.6` do not match.
38
38
  */
39
39
  export declare function isGpt6FamilyModel(model: unknown): boolean;
40
- /** Astra rejects `none` / `minimal`; OpenAI's migration floor is `low`. */
40
+ /** Astra/Sol reject `none` / `minimal`; OpenAI's migration floor is `low`. */
41
41
  export declare function coerceGpt6ReasoningEffort(model: unknown, effort: ThinkLevel | undefined): ThinkLevel | undefined;
42
+ /**
43
+ * GPT-6 Luna slugs (`gpt-6-luna`, `openai/gpt-6-luna`, `codex,gpt-6-luna`).
44
+ * Anchored so `gpt-6-sol` / `gpt-6-astra` do not match.
45
+ */
46
+ export declare function isGpt6LunaModel(model: unknown): boolean;
42
47
  /**
43
48
  * Remap unsupported GPT-6 efforts on a Responses/Unified request in place.
44
49
  * Covers convert (`openai-responses`) and same-protocol wire-keep (`codex`).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@caeliq/llms",
3
- "version": "1.0.69",
3
+ "version": "1.0.70",
4
4
  "description": "A universal LLM API transformation server",
5
5
  "main": "dist/cjs/server.cjs",
6
6
  "module": "dist/esm/server.mjs",
@@ -30,8 +30,8 @@
30
30
  ],
31
31
  "dependencies": {
32
32
  "@anthropic-ai/sdk": "^0.120.0",
33
- "@caeliq/ccr-shared": "^2.1.11",
34
- "@cursor/sdk": "^1.0.30",
33
+ "@caeliq/ccr-shared": "^2.1.12",
34
+ "@cursor/sdk": "^1.0.32",
35
35
  "@fastify/cors": "^11.3.0",
36
36
  "@fastify/rate-limit": "^11.2.0",
37
37
  "@google/genai": "^2.18.0",