@caeliq/llms 1.0.68 → 1.0.70

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/README.md +7 -5
  2. package/dist/cjs/server.cjs +223 -213
  3. package/dist/cjs/server.cjs.map +4 -4
  4. package/dist/esm/server.mjs +227 -217
  5. package/dist/esm/server.mjs.map +4 -4
  6. package/dist/routing/fim-pipeline.d.ts +6 -0
  7. package/dist/routing/protocol-endpoints.d.ts +1 -1
  8. package/dist/tests/fim.endpoint.d.ts +1 -0
  9. package/dist/tests/fim.transformers.d.ts +1 -0
  10. package/dist/tests/gpt6.astra-hardening.d.ts +1 -0
  11. package/dist/tests/opencode-gate-stubs.d.ts +1 -0
  12. package/dist/transformer/claude-auth.transformer.d.ts +1 -1
  13. package/dist/transformer/fim/fim.deepseek.transformer.d.ts +15 -0
  14. package/dist/transformer/fim/fim.mistral.transformer.d.ts +14 -0
  15. package/dist/transformer/fim/fim.qwen.transformer.d.ts +15 -0
  16. package/dist/transformer/fim/fim.transformer.d.ts +22 -0
  17. package/dist/transformer/fim/index.d.ts +4 -0
  18. package/dist/transformer/index.d.ts +5 -0
  19. package/dist/transformer/openai.transformer.d.ts +5 -0
  20. package/dist/transformer/opencode-headers.transformer.d.ts +92 -9
  21. package/dist/utils/claude-billing.d.ts +1 -1
  22. package/dist/utils/claude-model-catalog.d.ts +16 -1
  23. package/dist/utils/fim/encode.d.ts +14 -0
  24. package/dist/utils/fim/inbound.d.ts +9 -0
  25. package/dist/utils/fim/index.d.ts +7 -0
  26. package/dist/utils/fim/kinds.d.ts +16 -0
  27. package/dist/utils/fim/response.d.ts +17 -0
  28. package/dist/utils/fim/types.d.ts +41 -0
  29. package/dist/utils/fim/url.d.ts +13 -0
  30. package/dist/utils/reasoning-effort.d.ts +28 -0
  31. package/dist/utils/router.d.ts +2 -1
  32. package/package.json +3 -3
@@ -0,0 +1,6 @@
1
+ import type { FastifyInstance, FastifyReply, FastifyRequest } from "fastify";
2
+ import type { Transformer } from "../types/transformer";
3
+ /**
4
+ * Dedicated FIM pipeline — does not call prepareInboundRequest / chat Unified.
5
+ */
6
+ export declare function handleFimEndpoint(req: FastifyRequest, reply: FastifyReply, fastify: FastifyInstance, _ownerTransformer: Transformer, routePath: string): Promise<undefined>;
@@ -4,7 +4,7 @@ import type { ResponsesCallIdMap } from "../utils/openai.responses.util";
4
4
  /**
5
5
  * Inbound client protocols supported by CCR's gateway lifecycle.
6
6
  */
7
- export type ClientProtocol = "anthropic_messages" | "openai_chat_completions" | "openai_responses";
7
+ export type ClientProtocol = "anthropic_messages" | "openai_chat_completions" | "openai_responses" | "openai_fim_completions";
8
8
  export interface AnthropicSourceRequestFields {
9
9
  metadata?: Record<string, unknown>;
10
10
  thinking?: Record<string, unknown>;
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1 @@
1
+ export {};
@@ -48,7 +48,7 @@ export declare function modelIdForRequestedOneMillionBeta(modelId: string | unde
48
48
  * post-build Anthropic body (not the Unified request) because
49
49
  * `buildAnthropicBody` may synthesize `thinking`/`output_config` itself.
50
50
  */
51
- export declare function applyClaudeModelCapabilityAdjustments(anthropicBody: Record<string, any>, entry: ClaudeModelCatalogEntry | undefined): void;
51
+ export declare function applyClaudeModelCapabilityAdjustments(anthropicBody: Record<string, any>, entry: ClaudeModelCatalogEntry | undefined, logger?: any): void;
52
52
  /** Test-only reset hook so session-id state doesn't leak across test cases. */
53
53
  export declare function __resetClaudeAuthTransformerStateForTests(): void;
54
54
  /** Synthesized Claude Code identity headers for the non-Claude-Code branch. */
@@ -0,0 +1,15 @@
1
+ import type { LLMProvider } from "../../types/llm";
2
+ import type { Transformer, TransformerContext } from "../../types/transformer";
3
+ import { type UnifiedFimRequest } from "../../utils/fim";
4
+ /**
5
+ * DeepSeek beta completions FIM outbound.
6
+ * Same-kind (future deepseek inbound): auth + URL only.
7
+ * Cross-family from Codestral Unified: prompt+suffix + 4K clamp + non-thinking.
8
+ */
9
+ export declare class FimDeepseekTransformer implements Transformer {
10
+ static TransformerName: string;
11
+ name: string;
12
+ logger?: any;
13
+ transformRequestIn(request: UnifiedFimRequest | any, provider: LLMProvider, context: TransformerContext): Promise<Record<string, any>>;
14
+ transformResponseOut(response: Response): Promise<Response>;
15
+ }
@@ -0,0 +1,14 @@
1
+ import type { LLMProvider } from "../../types/llm";
2
+ import type { Transformer, TransformerContext } from "../../types/transformer";
3
+ import { type UnifiedFimRequest } from "../../utils/fim";
4
+ /**
5
+ * Codestral / Mistral native FIM outbound.
6
+ * Same-kind (mistral inbound): body passthrough — auth + URL only.
7
+ */
8
+ export declare class FimMistralTransformer implements Transformer {
9
+ static TransformerName: string;
10
+ name: string;
11
+ logger?: any;
12
+ transformRequestIn(request: UnifiedFimRequest | any, provider: LLMProvider, context: TransformerContext): Promise<Record<string, any>>;
13
+ transformResponseOut(response: Response): Promise<Response>;
14
+ }
@@ -0,0 +1,15 @@
1
+ import type { LLMProvider } from "../../types/llm";
2
+ import type { Transformer, TransformerContext } from "../../types/transformer";
3
+ import { type UnifiedFimRequest } from "../../utils/fim";
4
+ /**
5
+ * Qwen Completions FIM (LM Studio + DashScope).
6
+ * Same-kind (future qwen inbound): auth + URL only — do not re-template.
7
+ * Cross-family from Codestral Unified: HF tokens in prompt, no suffix field.
8
+ */
9
+ export declare class FimQwenTransformer implements Transformer {
10
+ static TransformerName: string;
11
+ name: string;
12
+ logger?: any;
13
+ transformRequestIn(request: UnifiedFimRequest | any, provider: LLMProvider, context: TransformerContext): Promise<Record<string, any>>;
14
+ transformResponseOut(response: Response): Promise<Response>;
15
+ }
@@ -0,0 +1,22 @@
1
+ import type { LLMProvider } from "../../types/llm";
2
+ import type { Transformer, TransformerContext } from "../../types/transformer";
3
+ import { type UnifiedFimRequest } from "../../utils/fim";
4
+ /**
5
+ * Protocol owner for POST /v1/fim/completions.
6
+ * Validates Codestral-shaped inbound → Unified FIM (v1).
7
+ * Client response framing: same-kind passthrough, else encode to inbound wire.
8
+ */
9
+ export declare class FimTransformer implements Transformer {
10
+ static TransformerName: string;
11
+ name: string;
12
+ endPoint: string;
13
+ logger?: any;
14
+ transformRequestOut(request: any, context: TransformerContext): Promise<any>;
15
+ /**
16
+ * Owner does not perform provider outbound; fim.* transformers do.
17
+ * transformResponseIn is a no-op identity for pipeline compatibility.
18
+ */
19
+ transformResponseIn(response: Response): Promise<Response>;
20
+ }
21
+ /** Type helper — FIM provider transformers accept UnifiedFimRequest. */
22
+ export type FimProviderTransformIn = (request: UnifiedFimRequest, provider: LLMProvider, context: TransformerContext) => Promise<Record<string, any>>;
@@ -0,0 +1,4 @@
1
+ export { FimTransformer } from "./fim.transformer";
2
+ export { FimMistralTransformer } from "./fim.mistral.transformer";
3
+ export { FimDeepseekTransformer } from "./fim.deepseek.transformer";
4
+ export { FimQwenTransformer } from "./fim.qwen.transformer";
@@ -30,6 +30,7 @@ import { ClaudeAuthTransformer } from "./claude-auth.transformer";
30
30
  import { CursorSdkTransformer } from "./cursor-sdk.transformer";
31
31
  import { AntigravityAuthTransformer } from "./antigravity-auth.transformer";
32
32
  import { XaiAuthTransformer } from "./xai-auth.transformer";
33
+ import { FimTransformer, FimMistralTransformer, FimDeepseekTransformer, FimQwenTransformer } from "./fim";
33
34
  declare const _default: {
34
35
  AnthropicTransformer: typeof AnthropicTransformer;
35
36
  GeminiTransformer: typeof GeminiTransformer;
@@ -63,5 +64,9 @@ declare const _default: {
63
64
  CursorSdkTransformer: typeof CursorSdkTransformer;
64
65
  AntigravityAuthTransformer: typeof AntigravityAuthTransformer;
65
66
  XaiAuthTransformer: typeof XaiAuthTransformer;
67
+ FimTransformer: typeof FimTransformer;
68
+ FimMistralTransformer: typeof FimMistralTransformer;
69
+ FimDeepseekTransformer: typeof FimDeepseekTransformer;
70
+ FimQwenTransformer: typeof FimQwenTransformer;
66
71
  };
67
72
  export default _default;
@@ -10,6 +10,8 @@ import { UnifiedChatRequest } from "../types/llm";
10
10
  *
11
11
  * ## Full request pipeline (for context)
12
12
  *
13
+ * Chat protocols share one lifecycle; the Anthropic inbound example:
14
+ *
13
15
  * Client → POST /v1/messages
14
16
  * → AnthropicTransformer.transformRequestOut() // Anthropic → Unified (OpenAI)
15
17
  * → provider.transformer.use[].transformRequestIn() // provider middleware
@@ -24,6 +26,9 @@ import { UnifiedChatRequest } from "../types/llm";
24
26
  * → OpenAITransformer.transformRequestOut() // validate → Unified
25
27
  * → provider.transformer.use[].transformRequestIn()
26
28
  * → … → OpenAITransformer.transformResponseIn() // reasoning_content; strip Unified thinking
29
+ *
30
+ * Responses and FIM use their own owners (`openai-responses`, `Fim`). FIM is a
31
+ * separate pipeline from chat Unified.
27
32
  */
28
33
  export declare class OpenAITransformer implements Transformer {
29
34
  name: string;
@@ -5,23 +5,37 @@ export declare class OpencodeHeadersTransformer implements Transformer {
5
5
  requestPhase: "transport";
6
6
  private lastTimestamp;
7
7
  private counter;
8
+ private firstEventTimeoutMs;
9
+ private firstProgressTimeoutMs;
10
+ private streamIdleTimeoutMs;
11
+ private reasoningIdleTimeoutMs;
8
12
  transformRequestIn(request: any, provider: any, context: any): Promise<Record<string, any>>;
9
- transformResponseOut(response: Response): Promise<Response>;
13
+ transformResponseOut(response: Response, context?: any): Promise<Response>;
14
+ /** Restore JSON for clients whose requests were streamed only to pass Zen's gate. */
15
+ private collectForcedStream;
10
16
  /**
11
- * Own the full upstream call so a `No provider available` 401 can be retried
12
- * with a fresh session in isolation. Any other non-ok response is re-thrown in
13
- * the exact shape sendRequestToProvider would have produced, so genuine
14
- * auth/rate/server errors keep flowing through the normal error + fallback
15
- * path unchanged.
17
+ * Own the full upstream call so Zen routing failures can be recovered by
18
+ * re-rolling the session, and transient failures retried on the same one.
19
+ * Exhausted routing failures become 503 so the normal fallback path can try
20
+ * another model; ordinary 4xx errors retain their upstream status.
16
21
  */
17
22
  private sendWithSessionRetry;
23
+ /**
24
+ * Zen can send SSE headers and lifecycle events, then stall before any
25
+ * output. Hold the response until a complete event and real output progress
26
+ * arrive while a fallback is still possible, then bound later idle reads.
27
+ * An open reasoning item emits nothing until it finishes when no summary is
28
+ * streamed, so while one is open the looser reasoning bound applies instead.
29
+ */
30
+ private requireFirstZenEvent;
18
31
  private preserveZenStreamErrors;
19
32
  private static zenStreamFailure;
20
33
  private buildHeaders;
21
34
  private resolveParentSessionId;
22
35
  /**
23
- * True only for the two Zen session-hash routing failures — a re-roll can
24
- * recover these. Kept deliberately narrow (exact status + message) so genuine
36
+ * True only for the two Zen session-hash routing failures — retried on the
37
+ * SAME session (which must not change during a conversation), then passed
38
+ * to fallback. Kept deliberately narrow (exact status + message) so genuine
25
39
  * auth errors (401 invalid key) and request errors (400 validation) are never
26
40
  * mistaken for routing failures and pass straight through.
27
41
  */
@@ -32,8 +46,77 @@ export declare class OpencodeHeadersTransformer implements Transformer {
32
46
  private retryAfterHeaders;
33
47
  private ensurePromptCacheKey;
34
48
  private ensurePromptCacheRetention;
49
+ /**
50
+ * Free-tier gate scope: `-free` models on a zen (opencode.ai) endpoint.
51
+ * Paid Zen models and non-Zen providers sharing this transformer skip the
52
+ * stub/stream handling below entirely.
53
+ */
54
+ private static isFreeTierZenRequest;
55
+ /**
56
+ * The free-tier gate answers `stream: false` with 403 FreeTierError even
57
+ * when everything else is exact (curl A/B 2026-09-24). Force SSE on the
58
+ * wire; the normal response path does not de-stream SSE for JSON clients,
59
+ * so transformRequestIn records the forced shape and transformResponseOut
60
+ * restores JSON via collectForcedStream.
61
+ */
62
+ private ensureStreamedForFreeTier;
63
+ /**
64
+ * Free-tier thinking arrives almost entirely as opaque `encrypted_content`;
65
+ * the only readable part is the reasoning summary, and Zen emits ~nothing
66
+ * unless `reasoning.summary` is asked for (curl A/B 2026-09-24, same prompt:
67
+ * `"detailed"` → 68 summary chars, `"auto"` → 0). Stamp `detailed` when the
68
+ * client already reasons but states no summary preference. An explicit
69
+ * client value (including `"none"`) always wins; absent/disabled reasoning
70
+ * is left alone so non-reasoning calls never gain a reasoning block.
71
+ */
72
+ private ensureDetailedSummaryForFreeTier;
73
+ /**
74
+ * Zen aborts free-tier Responses runs (`response.incomplete`, no further
75
+ * events — a client-side stall) unless `prompt_cache_key` equals the
76
+ * `x-opencode-session` header (curl A/B 2026-09-24). CCR's generic cache
77
+ * key (`ccr_<sha256>`, stable per conversation but foreign to Zen) must be
78
+ * replaced after `ensurePromptCacheKey` runs. Scoped to free-tier Zen on the
79
+ * Responses wire; chat bodies and paid models keep existing behavior.
80
+ * Runs per attempt so header and key stay in lockstep.
81
+ */
82
+ private applyFreeTierCacheKey;
83
+ /**
84
+ * Inject the exact-name `read`/`shell` function stubs the free-tier gate
85
+ * requires (curl A/B 2026-09-24: lowercase exact match; `Read`/`Bash`
86
+ * fail, schemas are free-form, extras harmless). Client tools are never
87
+ * modified or reordered; stubs are appended. Each stub clones its client
88
+ * counterpart's description/parameters when present so a
89
+ * stub call maps back onto a schema the client already accepts; otherwise
90
+ * it carries a minimal empty-object schema. Returns the alias map
91
+ * (stub -> client name, or null without counterpart) for the response
92
+ * stage; no aliases when nothing was injected.
93
+ */
94
+ private ensureGateStubTools;
95
+ private static readonly GATE_STUB_COUNTERPARTS;
96
+ private static outgoingToolShape;
97
+ private static toolName;
98
+ private static toolDef;
99
+ private static buildStubTool;
100
+ private static gateAliasesFrom;
101
+ /**
102
+ * Map gate-stub calls back to the advertised client tool inside SSE events.
103
+ * Only `name` fields on
104
+ * Responses `function_call` objects and chat `tool_calls[].function` objects
105
+ * are touched; text payloads and unrelated events pass through byte-identical.
106
+ * Stubs without a client counterpart are left alone (the client errors on
107
+ * them exactly as it would on any unknown tool).
108
+ */
109
+ private rewriteGateStubCalls;
110
+ private static renameGateStubCalls;
35
111
  private fingerprintConversation;
36
- private getOrCreateSessionId;
37
112
  private invalidateSession;
113
+ /**
114
+ * Conversation identity for the Zen session binding: an explicit client
115
+ * session id wherever the client supplies one (router-parsed, protocol
116
+ * context, or the shared header/body extractor used for cache keys), and
117
+ * the content fingerprint only for fully anonymous clients.
118
+ */
119
+ private resolveConversationId;
120
+ private getOrCreateSessionId;
38
121
  private generateId;
39
122
  }
@@ -9,7 +9,7 @@ export declare function extractFirstUserMessageText(messages: UnifiedMessage[] |
9
9
  export declare function computeVersionSuffix(text: string, version: string): string;
10
10
  /**
11
11
  * Current first-party Anthropic requests use the literal cch marker. Older
12
- * captures showed a random session value; the current 2.1.226 decompilation
12
+ * captures showed a random session value; the current 2.1.280 decompilation
13
13
  * gates this field to first-party/Vertex and emits `00000`.
14
14
  */
15
15
  export declare function sessionCch(): string;
@@ -1,6 +1,6 @@
1
1
  /**
2
2
  * Claude Code's bundled model capability catalog, revalidated against the
3
- * installed v2.1.226 CLI decompilation.
3
+ * installed v2.1.280 CLI decompilation.
4
4
  *
5
5
  * Every model-dependent decision in the claude-auth impersonation path (beta
6
6
  * flags, effort support, thinking shape, max_tokens ceiling) is driven by
@@ -21,6 +21,21 @@ export interface ClaudeModelCatalogEntry {
21
21
  };
22
22
  defaultEffort?: string;
23
23
  capabilities: string[];
24
+ /**
25
+ * CCR-owned Messages API request constraints. Not part of Claude Code's
26
+ * catalog (Claude Code simply never sends these shapes); sourced from
27
+ * Anthropic's per-model API documentation. Each flag names a request shape
28
+ * the model rejects with HTTP 400, so third-party emulation must normalize
29
+ * it away before sending.
30
+ */
31
+ apiConstraints?: {
32
+ /** `thinking: {type: "disabled"}` is rejected; thinking is always on. */
33
+ thinkingAlwaysOn?: boolean;
34
+ /** Forced `tool_choice` (`any` / `tool`) is rejected. */
35
+ noForcedToolChoice?: boolean;
36
+ /** `temperature` / `top_p` / `top_k` are rejected. */
37
+ noSamplingParams?: boolean;
38
+ };
24
39
  }
25
40
  export declare const CLAUDE_MODEL_CATALOG: Record<string, ClaudeModelCatalogEntry>;
26
41
  /** Strip the "[1m]" wire marker, reporting whether it was present. */
@@ -0,0 +1,14 @@
1
+ import type { UnifiedFimRequest } from "./types";
2
+ declare const DEEPSEEK_FIM_MAX_TOKENS = 4096;
3
+ /** Qwen / HF FIM markers for completions prompt. */
4
+ export declare function buildQwenFimPrompt(prompt: string, suffix?: string): string;
5
+ /** Sampling fields shared across FIM outbound bodies. */
6
+ export declare function pickFimSamplingFields(unified: UnifiedFimRequest): Record<string, unknown>;
7
+ /** Native prompt+suffix body (Mistral / DeepSeek field shape). */
8
+ export declare function encodePromptSuffixBody(unified: UnifiedFimRequest, options?: {
9
+ clampMaxTokens?: number;
10
+ disableThinking?: boolean;
11
+ }): Record<string, unknown>;
12
+ export declare function encodeDeepseekFimBody(unified: UnifiedFimRequest): Record<string, unknown>;
13
+ export declare function encodeQwenFimBody(unified: UnifiedFimRequest): Record<string, unknown>;
14
+ export { DEEPSEEK_FIM_MAX_TOKENS };
@@ -0,0 +1,9 @@
1
+ import type { FimInboundKind } from "./kinds";
2
+ import type { UnifiedFimRequest } from "./types";
3
+ /**
4
+ * Inbound → Unified FIM seam.
5
+ * v1: only Codestral/Mistral kind. Future: inboundKind selects an adapter.
6
+ */
7
+ export declare function inboundToUnifiedFim(body: unknown, inboundKind?: FimInboundKind): UnifiedFimRequest;
8
+ /** Clone client body for same-kind passthrough (preserve unknown fields). */
9
+ export declare function cloneFimClientBody(body: unknown, modelName: string): Record<string, unknown>;
@@ -0,0 +1,7 @@
1
+ export type { UnifiedFimRequest, UnifiedFimResponse, UnifiedFimChoice, } from "./types";
2
+ export type { FimInboundKind, FimOutboundFamily } from "./kinds";
3
+ export { V1_FIM_INBOUND_KIND, isFimProviderTransformerName, outboundFamilyFromTransformerName, shouldFimPassthrough, } from "./kinds";
4
+ export { inboundToUnifiedFim, cloneFimClientBody, } from "./inbound";
5
+ export { resolveFimMistralUrl, resolveFimDeepseekUrl, resolveFimQwenCompletionsUrl, bearerAuthHeaders, } from "./url";
6
+ export { buildQwenFimPrompt, pickFimSamplingFields, encodePromptSuffixBody, encodeDeepseekFimBody, encodeQwenFimBody, DEEPSEEK_FIM_MAX_TOKENS, } from "./encode";
7
+ export { encodeFimResponseForInbound, normalizeToFimClientJson, normalizeFimSseDataPayload, } from "./response";
@@ -0,0 +1,16 @@
1
+ /**
2
+ * FIM inbound/outbound family kinds. v1 only implements mistral/codestral
3
+ * inbound; deepseek/qwen inbound adapters are reserved for later.
4
+ */
5
+ export type FimInboundKind = "mistral" | "deepseek" | "qwen";
6
+ export type FimOutboundFamily = "mistral" | "deepseek" | "qwen";
7
+ /** v1 hard-wired inbound kind (Codestral/Mistral-shaped client wire). */
8
+ export declare const V1_FIM_INBOUND_KIND: FimInboundKind;
9
+ export declare function isFimProviderTransformerName(name: string | undefined): boolean;
10
+ export declare function outboundFamilyFromTransformerName(name: string | undefined): FimOutboundFamily | null;
11
+ /**
12
+ * Same-kind passthrough: no request or response shape translation —
13
+ * auth/URL only. Kept generic so future DeepSeek→DeepSeek / Qwen→Qwen
14
+ * inbound works the same.
15
+ */
16
+ export declare function shouldFimPassthrough(inboundKind: FimInboundKind, outboundFamily: FimOutboundFamily): boolean;
@@ -0,0 +1,17 @@
1
+ /**
2
+ * Encode upstream FIM responses to the **inbound** client wire.
3
+ * Client response shape follows inbound kind (not outbound provider).
4
+ * v1 inbound is mistral/Codestral; deepseek/qwen inbound reserved for later.
5
+ */
6
+ import type { FimInboundKind } from "./kinds";
7
+ /**
8
+ * Normalize upstream JSON to the inbound client wire.
9
+ * @param inboundKind — must match the kind used for inboundToUnifiedFim
10
+ */
11
+ export declare function encodeFimResponseForInbound(payload: any, inboundKind?: FimInboundKind): Record<string, any>;
12
+ /** @deprecated Prefer encodeFimResponseForInbound(payload, inboundKind) */
13
+ export declare function normalizeToFimClientJson(payload: any, inboundKind?: FimInboundKind): Record<string, any>;
14
+ /**
15
+ * Map one SSE data payload to the inbound client wire.
16
+ */
17
+ export declare function normalizeFimSseDataPayload(dataStr: string, inboundKind?: FimInboundKind): string;
@@ -0,0 +1,41 @@
1
+ /**
2
+ * Unified FIM intermediate (prompt + optional suffix + sampling).
3
+ * Types live next to FIM utils — not a separate types/fim package.
4
+ *
5
+ * Client response wire follows **inbound** kind (see encodeFimResponseForInbound).
6
+ * v1 inbound is mistral/Codestral chat.completion; other kinds reserved.
7
+ */
8
+ export interface UnifiedFimRequest {
9
+ model: string;
10
+ prompt: string;
11
+ suffix?: string;
12
+ max_tokens?: number;
13
+ temperature?: number;
14
+ top_p?: number;
15
+ stop?: string | string[];
16
+ stream?: boolean;
17
+ min_tokens?: number;
18
+ random_seed?: number;
19
+ }
20
+ /** Choice on mistral/Codestral inbound client wire. */
21
+ export interface UnifiedFimChoice {
22
+ index: number;
23
+ message: {
24
+ role: string;
25
+ content: string;
26
+ };
27
+ finish_reason: string;
28
+ }
29
+ export interface UnifiedFimResponse {
30
+ id: string;
31
+ object: "chat.completion";
32
+ model: string;
33
+ created: number;
34
+ choices: UnifiedFimChoice[];
35
+ usage: {
36
+ prompt_tokens: number;
37
+ completion_tokens: number;
38
+ total_tokens: number;
39
+ };
40
+ [key: string]: unknown;
41
+ }
@@ -0,0 +1,13 @@
1
+ /** Resolve Codestral/Mistral native FIM URL from provider base. */
2
+ export declare function resolveFimMistralUrl(baseUrl: string): URL;
3
+ /**
4
+ * DeepSeek hosted FIM uses beta completions.
5
+ * Prefer explicit …/beta/completions; otherwise derive from host.
6
+ */
7
+ export declare function resolveFimDeepseekUrl(baseUrl: string): URL;
8
+ /**
9
+ * Qwen / LM Studio / DashScope: OpenAI legacy completions.
10
+ * Trust api_base_url when it already ends with /completions.
11
+ */
12
+ export declare function resolveFimQwenCompletionsUrl(baseUrl: string): URL;
13
+ export declare function bearerAuthHeaders(apiKey: string | undefined): Record<string, string>;
@@ -32,4 +32,32 @@ export declare function canonicalReasoning(effortValue: unknown, enabledWhenEffo
32
32
  export declare function applyOpenAIChatReasoning(request: UnifiedChatRequest): UnifiedChatRequest;
33
33
  /** Anthropic accepts low..max, while CCR/OpenAI may additionally emit minimal/ultra/none. */
34
34
  export declare function toAnthropicReasoningEffort(effortValue: unknown): Exclude<ThinkLevel, "none" | "minimal" | "ultra"> | undefined;
35
+ /**
36
+ * GPT-6 family slugs (`gpt-6`, `gpt-6-astra`, `openai/gpt-6-astra`,
37
+ * `codex,gpt-6-astra`). Anchored so `gpt-60` / `gpt-5.6` do not match.
38
+ */
39
+ export declare function isGpt6FamilyModel(model: unknown): boolean;
40
+ /** Astra/Sol reject `none` / `minimal`; OpenAI's migration floor is `low`. */
41
+ export declare function coerceGpt6ReasoningEffort(model: unknown, effort: ThinkLevel | undefined): ThinkLevel | undefined;
42
+ /**
43
+ * GPT-6 Luna slugs (`gpt-6-luna`, `openai/gpt-6-luna`, `codex,gpt-6-luna`).
44
+ * Anchored so `gpt-6-sol` / `gpt-6-astra` do not match.
45
+ */
46
+ export declare function isGpt6LunaModel(model: unknown): boolean;
47
+ /**
48
+ * Remap unsupported GPT-6 efforts on a Responses/Unified request in place.
49
+ * Covers convert (`openai-responses`) and same-protocol wire-keep (`codex`).
50
+ */
51
+ export declare function applyGpt6ReasoningEffortCoercion(request: {
52
+ model?: unknown;
53
+ reasoning?: {
54
+ effort?: unknown;
55
+ enabled?: boolean;
56
+ } | null;
57
+ }): void;
58
+ /**
59
+ * GPT-6 Astra rejects temperature / top_p / logprobs on Responses. Strip in
60
+ * place when the model is in the gpt-6 family.
61
+ */
62
+ export declare function stripGpt6UnsupportedSampling(request: Record<string, any>): void;
35
63
  export {};
@@ -35,7 +35,7 @@ export interface RouterContext {
35
35
  tokenizerService?: TokenizerService;
36
36
  event?: any;
37
37
  }
38
- export type RouterScenarioType = 'default' | 'background' | 'think' | 'longContext' | 'webSearch' | 'subagent';
38
+ export type RouterScenarioType = 'default' | 'background' | 'think' | 'longContext' | 'webSearch' | 'subagent' | 'fim';
39
39
  export interface RouterFallbackConfig {
40
40
  default?: string[];
41
41
  background?: string[];
@@ -43,6 +43,7 @@ export interface RouterFallbackConfig {
43
43
  longContext?: string[];
44
44
  webSearch?: string[];
45
45
  subagent?: string[];
46
+ fim?: string[];
46
47
  }
47
48
  export declare const router: (req: any, _res: any, context: RouterContext) => Promise<void>;
48
49
  export declare const searchProjectBySession: (sessionId: string, logger?: any) => Promise<string | null>;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@caeliq/llms",
3
- "version": "1.0.68",
3
+ "version": "1.0.70",
4
4
  "description": "A universal LLM API transformation server",
5
5
  "main": "dist/cjs/server.cjs",
6
6
  "module": "dist/esm/server.mjs",
@@ -30,8 +30,8 @@
30
30
  ],
31
31
  "dependencies": {
32
32
  "@anthropic-ai/sdk": "^0.120.0",
33
- "@caeliq/ccr-shared": "^2.1.10",
34
- "@cursor/sdk": "^1.0.30",
33
+ "@caeliq/ccr-shared": "^2.1.12",
34
+ "@cursor/sdk": "^1.0.32",
35
35
  "@fastify/cors": "^11.3.0",
36
36
  "@fastify/rate-limit": "^11.2.0",
37
37
  "@google/genai": "^2.18.0",