@caeliq/llms 1.0.68 → 1.0.70
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -5
- package/dist/cjs/server.cjs +223 -213
- package/dist/cjs/server.cjs.map +4 -4
- package/dist/esm/server.mjs +227 -217
- package/dist/esm/server.mjs.map +4 -4
- package/dist/routing/fim-pipeline.d.ts +6 -0
- package/dist/routing/protocol-endpoints.d.ts +1 -1
- package/dist/tests/fim.endpoint.d.ts +1 -0
- package/dist/tests/fim.transformers.d.ts +1 -0
- package/dist/tests/gpt6.astra-hardening.d.ts +1 -0
- package/dist/tests/opencode-gate-stubs.d.ts +1 -0
- package/dist/transformer/claude-auth.transformer.d.ts +1 -1
- package/dist/transformer/fim/fim.deepseek.transformer.d.ts +15 -0
- package/dist/transformer/fim/fim.mistral.transformer.d.ts +14 -0
- package/dist/transformer/fim/fim.qwen.transformer.d.ts +15 -0
- package/dist/transformer/fim/fim.transformer.d.ts +22 -0
- package/dist/transformer/fim/index.d.ts +4 -0
- package/dist/transformer/index.d.ts +5 -0
- package/dist/transformer/openai.transformer.d.ts +5 -0
- package/dist/transformer/opencode-headers.transformer.d.ts +92 -9
- package/dist/utils/claude-billing.d.ts +1 -1
- package/dist/utils/claude-model-catalog.d.ts +16 -1
- package/dist/utils/fim/encode.d.ts +14 -0
- package/dist/utils/fim/inbound.d.ts +9 -0
- package/dist/utils/fim/index.d.ts +7 -0
- package/dist/utils/fim/kinds.d.ts +16 -0
- package/dist/utils/fim/response.d.ts +17 -0
- package/dist/utils/fim/types.d.ts +41 -0
- package/dist/utils/fim/url.d.ts +13 -0
- package/dist/utils/reasoning-effort.d.ts +28 -0
- package/dist/utils/router.d.ts +2 -1
- package/package.json +3 -3
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import type { FastifyInstance, FastifyReply, FastifyRequest } from "fastify";
|
|
2
|
+
import type { Transformer } from "../types/transformer";
|
|
3
|
+
/**
|
|
4
|
+
* Dedicated FIM pipeline — does not call prepareInboundRequest / chat Unified.
|
|
5
|
+
*/
|
|
6
|
+
export declare function handleFimEndpoint(req: FastifyRequest, reply: FastifyReply, fastify: FastifyInstance, _ownerTransformer: Transformer, routePath: string): Promise<undefined>;
|
|
@@ -4,7 +4,7 @@ import type { ResponsesCallIdMap } from "../utils/openai.responses.util";
|
|
|
4
4
|
/**
|
|
5
5
|
* Inbound client protocols supported by CCR's gateway lifecycle.
|
|
6
6
|
*/
|
|
7
|
-
export type ClientProtocol = "anthropic_messages" | "openai_chat_completions" | "openai_responses";
|
|
7
|
+
export type ClientProtocol = "anthropic_messages" | "openai_chat_completions" | "openai_responses" | "openai_fim_completions";
|
|
8
8
|
export interface AnthropicSourceRequestFields {
|
|
9
9
|
metadata?: Record<string, unknown>;
|
|
10
10
|
thinking?: Record<string, unknown>;
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -48,7 +48,7 @@ export declare function modelIdForRequestedOneMillionBeta(modelId: string | unde
|
|
|
48
48
|
* post-build Anthropic body (not the Unified request) because
|
|
49
49
|
* `buildAnthropicBody` may synthesize `thinking`/`output_config` itself.
|
|
50
50
|
*/
|
|
51
|
-
export declare function applyClaudeModelCapabilityAdjustments(anthropicBody: Record<string, any>, entry: ClaudeModelCatalogEntry | undefined): void;
|
|
51
|
+
export declare function applyClaudeModelCapabilityAdjustments(anthropicBody: Record<string, any>, entry: ClaudeModelCatalogEntry | undefined, logger?: any): void;
|
|
52
52
|
/** Test-only reset hook so session-id state doesn't leak across test cases. */
|
|
53
53
|
export declare function __resetClaudeAuthTransformerStateForTests(): void;
|
|
54
54
|
/** Synthesized Claude Code identity headers for the non-Claude-Code branch. */
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import type { LLMProvider } from "../../types/llm";
|
|
2
|
+
import type { Transformer, TransformerContext } from "../../types/transformer";
|
|
3
|
+
import { type UnifiedFimRequest } from "../../utils/fim";
|
|
4
|
+
/**
|
|
5
|
+
* DeepSeek beta completions FIM outbound.
|
|
6
|
+
* Same-kind (future deepseek inbound): auth + URL only.
|
|
7
|
+
* Cross-family from Codestral Unified: prompt+suffix + 4K clamp + non-thinking.
|
|
8
|
+
*/
|
|
9
|
+
export declare class FimDeepseekTransformer implements Transformer {
|
|
10
|
+
static TransformerName: string;
|
|
11
|
+
name: string;
|
|
12
|
+
logger?: any;
|
|
13
|
+
transformRequestIn(request: UnifiedFimRequest | any, provider: LLMProvider, context: TransformerContext): Promise<Record<string, any>>;
|
|
14
|
+
transformResponseOut(response: Response): Promise<Response>;
|
|
15
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import type { LLMProvider } from "../../types/llm";
|
|
2
|
+
import type { Transformer, TransformerContext } from "../../types/transformer";
|
|
3
|
+
import { type UnifiedFimRequest } from "../../utils/fim";
|
|
4
|
+
/**
|
|
5
|
+
* Codestral / Mistral native FIM outbound.
|
|
6
|
+
* Same-kind (mistral inbound): body passthrough — auth + URL only.
|
|
7
|
+
*/
|
|
8
|
+
export declare class FimMistralTransformer implements Transformer {
|
|
9
|
+
static TransformerName: string;
|
|
10
|
+
name: string;
|
|
11
|
+
logger?: any;
|
|
12
|
+
transformRequestIn(request: UnifiedFimRequest | any, provider: LLMProvider, context: TransformerContext): Promise<Record<string, any>>;
|
|
13
|
+
transformResponseOut(response: Response): Promise<Response>;
|
|
14
|
+
}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import type { LLMProvider } from "../../types/llm";
|
|
2
|
+
import type { Transformer, TransformerContext } from "../../types/transformer";
|
|
3
|
+
import { type UnifiedFimRequest } from "../../utils/fim";
|
|
4
|
+
/**
|
|
5
|
+
* Qwen Completions FIM (LM Studio + DashScope).
|
|
6
|
+
* Same-kind (future qwen inbound): auth + URL only — do not re-template.
|
|
7
|
+
* Cross-family from Codestral Unified: HF tokens in prompt, no suffix field.
|
|
8
|
+
*/
|
|
9
|
+
export declare class FimQwenTransformer implements Transformer {
|
|
10
|
+
static TransformerName: string;
|
|
11
|
+
name: string;
|
|
12
|
+
logger?: any;
|
|
13
|
+
transformRequestIn(request: UnifiedFimRequest | any, provider: LLMProvider, context: TransformerContext): Promise<Record<string, any>>;
|
|
14
|
+
transformResponseOut(response: Response): Promise<Response>;
|
|
15
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import type { LLMProvider } from "../../types/llm";
|
|
2
|
+
import type { Transformer, TransformerContext } from "../../types/transformer";
|
|
3
|
+
import { type UnifiedFimRequest } from "../../utils/fim";
|
|
4
|
+
/**
|
|
5
|
+
* Protocol owner for POST /v1/fim/completions.
|
|
6
|
+
* Validates Codestral-shaped inbound → Unified FIM (v1).
|
|
7
|
+
* Client response framing: same-kind passthrough, else encode to inbound wire.
|
|
8
|
+
*/
|
|
9
|
+
export declare class FimTransformer implements Transformer {
|
|
10
|
+
static TransformerName: string;
|
|
11
|
+
name: string;
|
|
12
|
+
endPoint: string;
|
|
13
|
+
logger?: any;
|
|
14
|
+
transformRequestOut(request: any, context: TransformerContext): Promise<any>;
|
|
15
|
+
/**
|
|
16
|
+
* Owner does not perform provider outbound; fim.* transformers do.
|
|
17
|
+
* transformResponseIn is a no-op identity for pipeline compatibility.
|
|
18
|
+
*/
|
|
19
|
+
transformResponseIn(response: Response): Promise<Response>;
|
|
20
|
+
}
|
|
21
|
+
/** Type helper — FIM provider transformers accept UnifiedFimRequest. */
|
|
22
|
+
export type FimProviderTransformIn = (request: UnifiedFimRequest, provider: LLMProvider, context: TransformerContext) => Promise<Record<string, any>>;
|
|
@@ -30,6 +30,7 @@ import { ClaudeAuthTransformer } from "./claude-auth.transformer";
|
|
|
30
30
|
import { CursorSdkTransformer } from "./cursor-sdk.transformer";
|
|
31
31
|
import { AntigravityAuthTransformer } from "./antigravity-auth.transformer";
|
|
32
32
|
import { XaiAuthTransformer } from "./xai-auth.transformer";
|
|
33
|
+
import { FimTransformer, FimMistralTransformer, FimDeepseekTransformer, FimQwenTransformer } from "./fim";
|
|
33
34
|
declare const _default: {
|
|
34
35
|
AnthropicTransformer: typeof AnthropicTransformer;
|
|
35
36
|
GeminiTransformer: typeof GeminiTransformer;
|
|
@@ -63,5 +64,9 @@ declare const _default: {
|
|
|
63
64
|
CursorSdkTransformer: typeof CursorSdkTransformer;
|
|
64
65
|
AntigravityAuthTransformer: typeof AntigravityAuthTransformer;
|
|
65
66
|
XaiAuthTransformer: typeof XaiAuthTransformer;
|
|
67
|
+
FimTransformer: typeof FimTransformer;
|
|
68
|
+
FimMistralTransformer: typeof FimMistralTransformer;
|
|
69
|
+
FimDeepseekTransformer: typeof FimDeepseekTransformer;
|
|
70
|
+
FimQwenTransformer: typeof FimQwenTransformer;
|
|
66
71
|
};
|
|
67
72
|
export default _default;
|
|
@@ -10,6 +10,8 @@ import { UnifiedChatRequest } from "../types/llm";
|
|
|
10
10
|
*
|
|
11
11
|
* ## Full request pipeline (for context)
|
|
12
12
|
*
|
|
13
|
+
* Chat protocols share one lifecycle; the Anthropic inbound example:
|
|
14
|
+
*
|
|
13
15
|
* Client → POST /v1/messages
|
|
14
16
|
* → AnthropicTransformer.transformRequestOut() // Anthropic → Unified (OpenAI)
|
|
15
17
|
* → provider.transformer.use[].transformRequestIn() // provider middleware
|
|
@@ -24,6 +26,9 @@ import { UnifiedChatRequest } from "../types/llm";
|
|
|
24
26
|
* → OpenAITransformer.transformRequestOut() // validate → Unified
|
|
25
27
|
* → provider.transformer.use[].transformRequestIn()
|
|
26
28
|
* → … → OpenAITransformer.transformResponseIn() // reasoning_content; strip Unified thinking
|
|
29
|
+
*
|
|
30
|
+
* Responses and FIM use their own owners (`openai-responses`, `Fim`). FIM is a
|
|
31
|
+
* separate pipeline from chat Unified.
|
|
27
32
|
*/
|
|
28
33
|
export declare class OpenAITransformer implements Transformer {
|
|
29
34
|
name: string;
|
|
@@ -5,23 +5,37 @@ export declare class OpencodeHeadersTransformer implements Transformer {
|
|
|
5
5
|
requestPhase: "transport";
|
|
6
6
|
private lastTimestamp;
|
|
7
7
|
private counter;
|
|
8
|
+
private firstEventTimeoutMs;
|
|
9
|
+
private firstProgressTimeoutMs;
|
|
10
|
+
private streamIdleTimeoutMs;
|
|
11
|
+
private reasoningIdleTimeoutMs;
|
|
8
12
|
transformRequestIn(request: any, provider: any, context: any): Promise<Record<string, any>>;
|
|
9
|
-
transformResponseOut(response: Response): Promise<Response>;
|
|
13
|
+
transformResponseOut(response: Response, context?: any): Promise<Response>;
|
|
14
|
+
/** Restore JSON for clients whose requests were streamed only to pass Zen's gate. */
|
|
15
|
+
private collectForcedStream;
|
|
10
16
|
/**
|
|
11
|
-
* Own the full upstream call so
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
* path unchanged.
|
|
17
|
+
* Own the full upstream call so Zen routing failures can be recovered by
|
|
18
|
+
* re-rolling the session, and transient failures retried on the same one.
|
|
19
|
+
* Exhausted routing failures become 503 so the normal fallback path can try
|
|
20
|
+
* another model; ordinary 4xx errors retain their upstream status.
|
|
16
21
|
*/
|
|
17
22
|
private sendWithSessionRetry;
|
|
23
|
+
/**
|
|
24
|
+
* Zen can send SSE headers and lifecycle events, then stall before any
|
|
25
|
+
* output. Hold the response until a complete event and real output progress
|
|
26
|
+
* arrive while a fallback is still possible, then bound later idle reads.
|
|
27
|
+
* An open reasoning item emits nothing until it finishes when no summary is
|
|
28
|
+
* streamed, so while one is open the looser reasoning bound applies instead.
|
|
29
|
+
*/
|
|
30
|
+
private requireFirstZenEvent;
|
|
18
31
|
private preserveZenStreamErrors;
|
|
19
32
|
private static zenStreamFailure;
|
|
20
33
|
private buildHeaders;
|
|
21
34
|
private resolveParentSessionId;
|
|
22
35
|
/**
|
|
23
|
-
* True only for the two Zen session-hash routing failures —
|
|
24
|
-
*
|
|
36
|
+
* True only for the two Zen session-hash routing failures — retried on the
|
|
37
|
+
* SAME session (which must not change during a conversation), then passed
|
|
38
|
+
* to fallback. Kept deliberately narrow (exact status + message) so genuine
|
|
25
39
|
* auth errors (401 invalid key) and request errors (400 validation) are never
|
|
26
40
|
* mistaken for routing failures and pass straight through.
|
|
27
41
|
*/
|
|
@@ -32,8 +46,77 @@ export declare class OpencodeHeadersTransformer implements Transformer {
|
|
|
32
46
|
private retryAfterHeaders;
|
|
33
47
|
private ensurePromptCacheKey;
|
|
34
48
|
private ensurePromptCacheRetention;
|
|
49
|
+
/**
|
|
50
|
+
* Free-tier gate scope: `-free` models on a zen (opencode.ai) endpoint.
|
|
51
|
+
* Paid Zen models and non-Zen providers sharing this transformer skip the
|
|
52
|
+
* stub/stream handling below entirely.
|
|
53
|
+
*/
|
|
54
|
+
private static isFreeTierZenRequest;
|
|
55
|
+
/**
|
|
56
|
+
* The free-tier gate answers `stream: false` with 403 FreeTierError even
|
|
57
|
+
* when everything else is exact (curl A/B 2026-09-24). Force SSE on the
|
|
58
|
+
* wire; the normal response path does not de-stream SSE for JSON clients,
|
|
59
|
+
* so transformRequestIn records the forced shape and transformResponseOut
|
|
60
|
+
* restores JSON via collectForcedStream.
|
|
61
|
+
*/
|
|
62
|
+
private ensureStreamedForFreeTier;
|
|
63
|
+
/**
|
|
64
|
+
* Free-tier thinking arrives almost entirely as opaque `encrypted_content`;
|
|
65
|
+
* the only readable part is the reasoning summary, and Zen emits ~nothing
|
|
66
|
+
* unless `reasoning.summary` is asked for (curl A/B 2026-09-24, same prompt:
|
|
67
|
+
* `"detailed"` → 68 summary chars, `"auto"` → 0). Stamp `detailed` when the
|
|
68
|
+
* client already reasons but states no summary preference. An explicit
|
|
69
|
+
* client value (including `"none"`) always wins; absent/disabled reasoning
|
|
70
|
+
* is left alone so non-reasoning calls never gain a reasoning block.
|
|
71
|
+
*/
|
|
72
|
+
private ensureDetailedSummaryForFreeTier;
|
|
73
|
+
/**
|
|
74
|
+
* Zen aborts free-tier Responses runs (`response.incomplete`, no further
|
|
75
|
+
* events — a client-side stall) unless `prompt_cache_key` equals the
|
|
76
|
+
* `x-opencode-session` header (curl A/B 2026-09-24). CCR's generic cache
|
|
77
|
+
* key (`ccr_<sha256>`, stable per conversation but foreign to Zen) must be
|
|
78
|
+
* replaced after `ensurePromptCacheKey` runs. Scoped to free-tier Zen on the
|
|
79
|
+
* Responses wire; chat bodies and paid models keep existing behavior.
|
|
80
|
+
* Runs per attempt so header and key stay in lockstep.
|
|
81
|
+
*/
|
|
82
|
+
private applyFreeTierCacheKey;
|
|
83
|
+
/**
|
|
84
|
+
* Inject the exact-name `read`/`shell` function stubs the free-tier gate
|
|
85
|
+
* requires (curl A/B 2026-09-24: lowercase exact match; `Read`/`Bash`
|
|
86
|
+
* fail, schemas are free-form, extras harmless). Client tools are never
|
|
87
|
+
* modified or reordered; stubs are appended. Each stub clones its client
|
|
88
|
+
* counterpart's description/parameters when present so a
|
|
89
|
+
* stub call maps back onto a schema the client already accepts; otherwise
|
|
90
|
+
* it carries a minimal empty-object schema. Returns the alias map
|
|
91
|
+
* (stub -> client name, or null without counterpart) for the response
|
|
92
|
+
* stage; no aliases when nothing was injected.
|
|
93
|
+
*/
|
|
94
|
+
private ensureGateStubTools;
|
|
95
|
+
private static readonly GATE_STUB_COUNTERPARTS;
|
|
96
|
+
private static outgoingToolShape;
|
|
97
|
+
private static toolName;
|
|
98
|
+
private static toolDef;
|
|
99
|
+
private static buildStubTool;
|
|
100
|
+
private static gateAliasesFrom;
|
|
101
|
+
/**
|
|
102
|
+
* Map gate-stub calls back to the advertised client tool inside SSE events.
|
|
103
|
+
* Only `name` fields on
|
|
104
|
+
* Responses `function_call` objects and chat `tool_calls[].function` objects
|
|
105
|
+
* are touched; text payloads and unrelated events pass through byte-identical.
|
|
106
|
+
* Stubs without a client counterpart are left alone (the client errors on
|
|
107
|
+
* them exactly as it would on any unknown tool).
|
|
108
|
+
*/
|
|
109
|
+
private rewriteGateStubCalls;
|
|
110
|
+
private static renameGateStubCalls;
|
|
35
111
|
private fingerprintConversation;
|
|
36
|
-
private getOrCreateSessionId;
|
|
37
112
|
private invalidateSession;
|
|
113
|
+
/**
|
|
114
|
+
* Conversation identity for the Zen session binding: an explicit client
|
|
115
|
+
* session id wherever the client supplies one (router-parsed, protocol
|
|
116
|
+
* context, or the shared header/body extractor used for cache keys), and
|
|
117
|
+
* the content fingerprint only for fully anonymous clients.
|
|
118
|
+
*/
|
|
119
|
+
private resolveConversationId;
|
|
120
|
+
private getOrCreateSessionId;
|
|
38
121
|
private generateId;
|
|
39
122
|
}
|
|
@@ -9,7 +9,7 @@ export declare function extractFirstUserMessageText(messages: UnifiedMessage[] |
|
|
|
9
9
|
export declare function computeVersionSuffix(text: string, version: string): string;
|
|
10
10
|
/**
|
|
11
11
|
* Current first-party Anthropic requests use the literal cch marker. Older
|
|
12
|
-
* captures showed a random session value; the current 2.1.
|
|
12
|
+
* captures showed a random session value; the current 2.1.280 decompilation
|
|
13
13
|
* gates this field to first-party/Vertex and emits `00000`.
|
|
14
14
|
*/
|
|
15
15
|
export declare function sessionCch(): string;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Claude Code's bundled model capability catalog, revalidated against the
|
|
3
|
-
* installed v2.1.
|
|
3
|
+
* installed v2.1.280 CLI decompilation.
|
|
4
4
|
*
|
|
5
5
|
* Every model-dependent decision in the claude-auth impersonation path (beta
|
|
6
6
|
* flags, effort support, thinking shape, max_tokens ceiling) is driven by
|
|
@@ -21,6 +21,21 @@ export interface ClaudeModelCatalogEntry {
|
|
|
21
21
|
};
|
|
22
22
|
defaultEffort?: string;
|
|
23
23
|
capabilities: string[];
|
|
24
|
+
/**
|
|
25
|
+
* CCR-owned Messages API request constraints. Not part of Claude Code's
|
|
26
|
+
* catalog (Claude Code simply never sends these shapes); sourced from
|
|
27
|
+
* Anthropic's per-model API documentation. Each flag names a request shape
|
|
28
|
+
* the model rejects with HTTP 400, so third-party emulation must normalize
|
|
29
|
+
* it away before sending.
|
|
30
|
+
*/
|
|
31
|
+
apiConstraints?: {
|
|
32
|
+
/** `thinking: {type: "disabled"}` is rejected; thinking is always on. */
|
|
33
|
+
thinkingAlwaysOn?: boolean;
|
|
34
|
+
/** Forced `tool_choice` (`any` / `tool`) is rejected. */
|
|
35
|
+
noForcedToolChoice?: boolean;
|
|
36
|
+
/** `temperature` / `top_p` / `top_k` are rejected. */
|
|
37
|
+
noSamplingParams?: boolean;
|
|
38
|
+
};
|
|
24
39
|
}
|
|
25
40
|
export declare const CLAUDE_MODEL_CATALOG: Record<string, ClaudeModelCatalogEntry>;
|
|
26
41
|
/** Strip the "[1m]" wire marker, reporting whether it was present. */
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import type { UnifiedFimRequest } from "./types";
|
|
2
|
+
declare const DEEPSEEK_FIM_MAX_TOKENS = 4096;
|
|
3
|
+
/** Qwen / HF FIM markers for completions prompt. */
|
|
4
|
+
export declare function buildQwenFimPrompt(prompt: string, suffix?: string): string;
|
|
5
|
+
/** Sampling fields shared across FIM outbound bodies. */
|
|
6
|
+
export declare function pickFimSamplingFields(unified: UnifiedFimRequest): Record<string, unknown>;
|
|
7
|
+
/** Native prompt+suffix body (Mistral / DeepSeek field shape). */
|
|
8
|
+
export declare function encodePromptSuffixBody(unified: UnifiedFimRequest, options?: {
|
|
9
|
+
clampMaxTokens?: number;
|
|
10
|
+
disableThinking?: boolean;
|
|
11
|
+
}): Record<string, unknown>;
|
|
12
|
+
export declare function encodeDeepseekFimBody(unified: UnifiedFimRequest): Record<string, unknown>;
|
|
13
|
+
export declare function encodeQwenFimBody(unified: UnifiedFimRequest): Record<string, unknown>;
|
|
14
|
+
export { DEEPSEEK_FIM_MAX_TOKENS };
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
import type { FimInboundKind } from "./kinds";
|
|
2
|
+
import type { UnifiedFimRequest } from "./types";
|
|
3
|
+
/**
|
|
4
|
+
* Inbound → Unified FIM seam.
|
|
5
|
+
* v1: only Codestral/Mistral kind. Future: inboundKind selects an adapter.
|
|
6
|
+
*/
|
|
7
|
+
export declare function inboundToUnifiedFim(body: unknown, inboundKind?: FimInboundKind): UnifiedFimRequest;
|
|
8
|
+
/** Clone client body for same-kind passthrough (preserve unknown fields). */
|
|
9
|
+
export declare function cloneFimClientBody(body: unknown, modelName: string): Record<string, unknown>;
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
export type { UnifiedFimRequest, UnifiedFimResponse, UnifiedFimChoice, } from "./types";
|
|
2
|
+
export type { FimInboundKind, FimOutboundFamily } from "./kinds";
|
|
3
|
+
export { V1_FIM_INBOUND_KIND, isFimProviderTransformerName, outboundFamilyFromTransformerName, shouldFimPassthrough, } from "./kinds";
|
|
4
|
+
export { inboundToUnifiedFim, cloneFimClientBody, } from "./inbound";
|
|
5
|
+
export { resolveFimMistralUrl, resolveFimDeepseekUrl, resolveFimQwenCompletionsUrl, bearerAuthHeaders, } from "./url";
|
|
6
|
+
export { buildQwenFimPrompt, pickFimSamplingFields, encodePromptSuffixBody, encodeDeepseekFimBody, encodeQwenFimBody, DEEPSEEK_FIM_MAX_TOKENS, } from "./encode";
|
|
7
|
+
export { encodeFimResponseForInbound, normalizeToFimClientJson, normalizeFimSseDataPayload, } from "./response";
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* FIM inbound/outbound family kinds. v1 only implements mistral/codestral
|
|
3
|
+
* inbound; deepseek/qwen inbound adapters are reserved for later.
|
|
4
|
+
*/
|
|
5
|
+
export type FimInboundKind = "mistral" | "deepseek" | "qwen";
|
|
6
|
+
export type FimOutboundFamily = "mistral" | "deepseek" | "qwen";
|
|
7
|
+
/** v1 hard-wired inbound kind (Codestral/Mistral-shaped client wire). */
|
|
8
|
+
export declare const V1_FIM_INBOUND_KIND: FimInboundKind;
|
|
9
|
+
export declare function isFimProviderTransformerName(name: string | undefined): boolean;
|
|
10
|
+
export declare function outboundFamilyFromTransformerName(name: string | undefined): FimOutboundFamily | null;
|
|
11
|
+
/**
|
|
12
|
+
* Same-kind passthrough: no request or response shape translation —
|
|
13
|
+
* auth/URL only. Kept generic so future DeepSeek→DeepSeek / Qwen→Qwen
|
|
14
|
+
* inbound works the same.
|
|
15
|
+
*/
|
|
16
|
+
export declare function shouldFimPassthrough(inboundKind: FimInboundKind, outboundFamily: FimOutboundFamily): boolean;
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Encode upstream FIM responses to the **inbound** client wire.
|
|
3
|
+
* Client response shape follows inbound kind (not outbound provider).
|
|
4
|
+
* v1 inbound is mistral/Codestral; deepseek/qwen inbound reserved for later.
|
|
5
|
+
*/
|
|
6
|
+
import type { FimInboundKind } from "./kinds";
|
|
7
|
+
/**
|
|
8
|
+
* Normalize upstream JSON to the inbound client wire.
|
|
9
|
+
* @param inboundKind — must match the kind used for inboundToUnifiedFim
|
|
10
|
+
*/
|
|
11
|
+
export declare function encodeFimResponseForInbound(payload: any, inboundKind?: FimInboundKind): Record<string, any>;
|
|
12
|
+
/** @deprecated Prefer encodeFimResponseForInbound(payload, inboundKind) */
|
|
13
|
+
export declare function normalizeToFimClientJson(payload: any, inboundKind?: FimInboundKind): Record<string, any>;
|
|
14
|
+
/**
|
|
15
|
+
* Map one SSE data payload to the inbound client wire.
|
|
16
|
+
*/
|
|
17
|
+
export declare function normalizeFimSseDataPayload(dataStr: string, inboundKind?: FimInboundKind): string;
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Unified FIM intermediate (prompt + optional suffix + sampling).
|
|
3
|
+
* Types live next to FIM utils — not a separate types/fim package.
|
|
4
|
+
*
|
|
5
|
+
* Client response wire follows **inbound** kind (see encodeFimResponseForInbound).
|
|
6
|
+
* v1 inbound is mistral/Codestral chat.completion; other kinds reserved.
|
|
7
|
+
*/
|
|
8
|
+
export interface UnifiedFimRequest {
|
|
9
|
+
model: string;
|
|
10
|
+
prompt: string;
|
|
11
|
+
suffix?: string;
|
|
12
|
+
max_tokens?: number;
|
|
13
|
+
temperature?: number;
|
|
14
|
+
top_p?: number;
|
|
15
|
+
stop?: string | string[];
|
|
16
|
+
stream?: boolean;
|
|
17
|
+
min_tokens?: number;
|
|
18
|
+
random_seed?: number;
|
|
19
|
+
}
|
|
20
|
+
/** Choice on mistral/Codestral inbound client wire. */
|
|
21
|
+
export interface UnifiedFimChoice {
|
|
22
|
+
index: number;
|
|
23
|
+
message: {
|
|
24
|
+
role: string;
|
|
25
|
+
content: string;
|
|
26
|
+
};
|
|
27
|
+
finish_reason: string;
|
|
28
|
+
}
|
|
29
|
+
export interface UnifiedFimResponse {
|
|
30
|
+
id: string;
|
|
31
|
+
object: "chat.completion";
|
|
32
|
+
model: string;
|
|
33
|
+
created: number;
|
|
34
|
+
choices: UnifiedFimChoice[];
|
|
35
|
+
usage: {
|
|
36
|
+
prompt_tokens: number;
|
|
37
|
+
completion_tokens: number;
|
|
38
|
+
total_tokens: number;
|
|
39
|
+
};
|
|
40
|
+
[key: string]: unknown;
|
|
41
|
+
}
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
/** Resolve Codestral/Mistral native FIM URL from provider base. */
|
|
2
|
+
export declare function resolveFimMistralUrl(baseUrl: string): URL;
|
|
3
|
+
/**
|
|
4
|
+
* DeepSeek hosted FIM uses beta completions.
|
|
5
|
+
* Prefer explicit …/beta/completions; otherwise derive from host.
|
|
6
|
+
*/
|
|
7
|
+
export declare function resolveFimDeepseekUrl(baseUrl: string): URL;
|
|
8
|
+
/**
|
|
9
|
+
* Qwen / LM Studio / DashScope: OpenAI legacy completions.
|
|
10
|
+
* Trust api_base_url when it already ends with /completions.
|
|
11
|
+
*/
|
|
12
|
+
export declare function resolveFimQwenCompletionsUrl(baseUrl: string): URL;
|
|
13
|
+
export declare function bearerAuthHeaders(apiKey: string | undefined): Record<string, string>;
|
|
@@ -32,4 +32,32 @@ export declare function canonicalReasoning(effortValue: unknown, enabledWhenEffo
|
|
|
32
32
|
export declare function applyOpenAIChatReasoning(request: UnifiedChatRequest): UnifiedChatRequest;
|
|
33
33
|
/** Anthropic accepts low..max, while CCR/OpenAI may additionally emit minimal/ultra/none. */
|
|
34
34
|
export declare function toAnthropicReasoningEffort(effortValue: unknown): Exclude<ThinkLevel, "none" | "minimal" | "ultra"> | undefined;
|
|
35
|
+
/**
|
|
36
|
+
* GPT-6 family slugs (`gpt-6`, `gpt-6-astra`, `openai/gpt-6-astra`,
|
|
37
|
+
* `codex,gpt-6-astra`). Anchored so `gpt-60` / `gpt-5.6` do not match.
|
|
38
|
+
*/
|
|
39
|
+
export declare function isGpt6FamilyModel(model: unknown): boolean;
|
|
40
|
+
/** Astra/Sol reject `none` / `minimal`; OpenAI's migration floor is `low`. */
|
|
41
|
+
export declare function coerceGpt6ReasoningEffort(model: unknown, effort: ThinkLevel | undefined): ThinkLevel | undefined;
|
|
42
|
+
/**
|
|
43
|
+
* GPT-6 Luna slugs (`gpt-6-luna`, `openai/gpt-6-luna`, `codex,gpt-6-luna`).
|
|
44
|
+
* Anchored so `gpt-6-sol` / `gpt-6-astra` do not match.
|
|
45
|
+
*/
|
|
46
|
+
export declare function isGpt6LunaModel(model: unknown): boolean;
|
|
47
|
+
/**
|
|
48
|
+
* Remap unsupported GPT-6 efforts on a Responses/Unified request in place.
|
|
49
|
+
* Covers convert (`openai-responses`) and same-protocol wire-keep (`codex`).
|
|
50
|
+
*/
|
|
51
|
+
export declare function applyGpt6ReasoningEffortCoercion(request: {
|
|
52
|
+
model?: unknown;
|
|
53
|
+
reasoning?: {
|
|
54
|
+
effort?: unknown;
|
|
55
|
+
enabled?: boolean;
|
|
56
|
+
} | null;
|
|
57
|
+
}): void;
|
|
58
|
+
/**
|
|
59
|
+
* GPT-6 Astra rejects temperature / top_p / logprobs on Responses. Strip in
|
|
60
|
+
* place when the model is in the gpt-6 family.
|
|
61
|
+
*/
|
|
62
|
+
export declare function stripGpt6UnsupportedSampling(request: Record<string, any>): void;
|
|
35
63
|
export {};
|
package/dist/utils/router.d.ts
CHANGED
|
@@ -35,7 +35,7 @@ export interface RouterContext {
|
|
|
35
35
|
tokenizerService?: TokenizerService;
|
|
36
36
|
event?: any;
|
|
37
37
|
}
|
|
38
|
-
export type RouterScenarioType = 'default' | 'background' | 'think' | 'longContext' | 'webSearch' | 'subagent';
|
|
38
|
+
export type RouterScenarioType = 'default' | 'background' | 'think' | 'longContext' | 'webSearch' | 'subagent' | 'fim';
|
|
39
39
|
export interface RouterFallbackConfig {
|
|
40
40
|
default?: string[];
|
|
41
41
|
background?: string[];
|
|
@@ -43,6 +43,7 @@ export interface RouterFallbackConfig {
|
|
|
43
43
|
longContext?: string[];
|
|
44
44
|
webSearch?: string[];
|
|
45
45
|
subagent?: string[];
|
|
46
|
+
fim?: string[];
|
|
46
47
|
}
|
|
47
48
|
export declare const router: (req: any, _res: any, context: RouterContext) => Promise<void>;
|
|
48
49
|
export declare const searchProjectBySession: (sessionId: string, logger?: any) => Promise<string | null>;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@caeliq/llms",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.70",
|
|
4
4
|
"description": "A universal LLM API transformation server",
|
|
5
5
|
"main": "dist/cjs/server.cjs",
|
|
6
6
|
"module": "dist/esm/server.mjs",
|
|
@@ -30,8 +30,8 @@
|
|
|
30
30
|
],
|
|
31
31
|
"dependencies": {
|
|
32
32
|
"@anthropic-ai/sdk": "^0.120.0",
|
|
33
|
-
"@caeliq/ccr-shared": "^2.1.
|
|
34
|
-
"@cursor/sdk": "^1.0.
|
|
33
|
+
"@caeliq/ccr-shared": "^2.1.12",
|
|
34
|
+
"@cursor/sdk": "^1.0.32",
|
|
35
35
|
"@fastify/cors": "^11.3.0",
|
|
36
36
|
"@fastify/rate-limit": "^11.2.0",
|
|
37
37
|
"@google/genai": "^2.18.0",
|