@caeliq/llms 1.0.72 → 1.0.73
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cjs/server.cjs +233 -230
- package/dist/cjs/server.cjs.map +4 -4
- package/dist/cursor-sdk/inbound-session.d.ts +21 -0
- package/dist/cursor-sdk/model-selection.d.ts +23 -0
- package/dist/cursor-sdk/session.d.ts +16 -5
- package/dist/cursor-sdk/shared.d.ts +10 -1
- package/dist/cursor-sdk/turn-output.d.ts +2 -0
- package/dist/esm/server.mjs +233 -230
- package/dist/esm/server.mjs.map +4 -4
- package/dist/services/provider.d.ts +8 -2
- package/dist/session-registry.d.ts +30 -3
- package/dist/tests/codex.bootstrap-buffer.d.ts +1 -0
- package/dist/tests/codex.model-catalog.d.ts +1 -0
- package/dist/tests/cursor-sdk.inbound-session.d.ts +1 -0
- package/dist/tests/cursor-sdk.interrupt-reentry.d.ts +1 -1
- package/dist/tests/cursor-sdk.model-remint.d.ts +1 -0
- package/dist/tests/cursor-sdk.model-selection.d.ts +1 -0
- package/dist/tests/cursor-sdk.on-delta-thinking.d.ts +1 -1
- package/dist/tests/cursor-sdk.runner-recovery.d.ts +1 -1
- package/dist/tests/cursor-sdk.scratch-report.d.ts +1 -1
- package/dist/tests/cursor-sdk.turn-coordination.d.ts +1 -1
- package/dist/tests/fallback.request-context.routes.d.ts +1 -0
- package/dist/tests/path-alias.build.d.ts +1 -0
- package/dist/tests/request-scoped-errors.d.ts +1 -0
- package/dist/tests/request-scoped-errors.routes.d.ts +1 -0
- package/dist/tests/responses.multi-agent.routes.d.ts +1 -0
- package/dist/tests/responses.orphan-delegation.d.ts +1 -0
- package/dist/tests/support/isolate-session-registry.d.ts +1 -0
- package/dist/transformer/codex.transformer.d.ts +18 -0
- package/dist/types/llm.d.ts +21 -2
- package/dist/utils/codex-bootstrap.d.ts +68 -0
- package/dist/utils/codex-model-catalog.d.ts +63 -0
- package/dist/utils/nested-agent.d.ts +4 -0
- package/dist/utils/openai.responses.util.d.ts +37 -1
- package/dist/utils/reasoning-effort.d.ts +3 -20
- package/dist/utils/request-scoped-errors.d.ts +92 -0
- package/package.json +3 -3
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { LLMProvider, RegisterProviderRequest, ModelRoute, RequestRouteInfo } from "../types/llm";
|
|
1
|
+
import { LLMProvider, RegisterProviderRequest, ModelRoute, ProviderUpdate, RequestRouteInfo } from "../types/llm";
|
|
2
2
|
import { ConfigService } from "./config";
|
|
3
3
|
import { TransformerService } from "./transformer";
|
|
4
4
|
export declare class ProviderService {
|
|
@@ -12,10 +12,16 @@ export declare class ProviderService {
|
|
|
12
12
|
private initializeFromProvidersArray;
|
|
13
13
|
/** Match TransformerService: instances used in provider use[] need the service logger. */
|
|
14
14
|
private attachTransformerLogger;
|
|
15
|
+
/**
|
|
16
|
+
* Store request-scoped error rules under the canonical key only, so an
|
|
17
|
+
* update under any spelling cannot be shadowed by a stale alias. Rules are
|
|
18
|
+
* validated here to surface config mistakes at registration time.
|
|
19
|
+
*/
|
|
20
|
+
private withCanonicalScopedErrorRules;
|
|
15
21
|
registerProvider(request: RegisterProviderRequest): LLMProvider;
|
|
16
22
|
getProviders(): LLMProvider[];
|
|
17
23
|
getProvider(name: string): LLMProvider | undefined;
|
|
18
|
-
updateProvider(id: string, updates:
|
|
24
|
+
updateProvider(id: string, updates: ProviderUpdate): LLMProvider | null;
|
|
19
25
|
deleteProvider(id: string): boolean;
|
|
20
26
|
toggleProvider(name: string, _enabled: boolean): boolean;
|
|
21
27
|
resolveModelRoute(modelName: string): RequestRouteInfo | null;
|
|
@@ -1,26 +1,42 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Persistent registry for every session identity CCR mints.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
4
|
+
* Several families, one mechanism — the stable lookup key already exists in each
|
|
5
5
|
* case (Zen conversation id, cursor buildSessionKey hash); only the minted
|
|
6
6
|
* value used to live in process memory and died on restart:
|
|
7
7
|
*
|
|
8
8
|
* - "zen": conversationKey -> x-opencode-session (ses_…)
|
|
9
9
|
* - "cursor": sessionKey -> { agentId, workspaceDir, model }
|
|
10
|
+
* - "cursor-inbound": hashed protocol conversation key -> internal Cursor id
|
|
11
|
+
* (cursor-sdk/inbound-session.ts; written on mint, claim, new user text,
|
|
12
|
+
* and at most hourly as a TTL refresh)
|
|
10
13
|
*
|
|
11
14
|
* The file lives under CCR_HOME (the mounted ~/.claude-code-router volume),
|
|
12
15
|
* so bindings survive both restarts and image rebuilds. Plain JSON: values
|
|
13
16
|
* are short strings, not blobs (unlike cursor-opencode-provider's pb.gz,
|
|
14
17
|
* which persists raw Cursor protocol state we never see — the SDK owns that).
|
|
15
18
|
*
|
|
16
|
-
* Synchronous API, tiny file (capped
|
|
17
|
-
* only — never on the
|
|
19
|
+
* Synchronous API, tiny file (entries capped per family), persistence on
|
|
20
|
+
* writes only — never on the read path.
|
|
18
21
|
*/
|
|
19
22
|
export type PersistedSession = {
|
|
20
23
|
/** The fixed CCR-minted id (ses_… for zen; SDK agentId for cursor). */
|
|
21
24
|
sessionId: string;
|
|
22
25
|
workspaceDir?: string;
|
|
23
26
|
model?: string;
|
|
27
|
+
/** `id|sorted params` from cursorModelFingerprint. Missing means pre-variant binding. */
|
|
28
|
+
modelFingerprint?: string;
|
|
29
|
+
/**
|
|
30
|
+
* Cursor inbound opening row: a follow-up turn claimed it. A repeated
|
|
31
|
+
* opening then needs a fresh id.
|
|
32
|
+
*/
|
|
33
|
+
progressed?: boolean;
|
|
34
|
+
/**
|
|
35
|
+
* Cursor inbound opening row that replaced an unclaimed opening of an
|
|
36
|
+
* identical earlier conversation. A follow-up cannot tell the two apart,
|
|
37
|
+
* so it gets a fresh agent instead of claiming this one.
|
|
38
|
+
*/
|
|
39
|
+
contested?: boolean;
|
|
24
40
|
updatedAt: number;
|
|
25
41
|
};
|
|
26
42
|
export declare const SESSION_REGISTRY_TTL_MS: number;
|
|
@@ -28,6 +44,17 @@ export declare const SESSION_REGISTRY_TTL_MS: number;
|
|
|
28
44
|
export declare function getPersistedSession(family: string, key: string, now?: number): PersistedSession | undefined;
|
|
29
45
|
/** Record a newly minted fixed id. Overwrites any prior binding. */
|
|
30
46
|
export declare function putPersistedSession(family: string, key: string, value: Omit<PersistedSession, "updatedAt">, now?: number): PersistedSession;
|
|
47
|
+
/**
|
|
48
|
+
* Apply several writes to one family with a single file rewrite: `put`
|
|
49
|
+
* entries are stored (overwriting), then `remove` keys are dropped.
|
|
50
|
+
*/
|
|
51
|
+
export declare function updatePersistedSessions(family: string, changes: {
|
|
52
|
+
put?: Array<{
|
|
53
|
+
key: string;
|
|
54
|
+
value: Omit<PersistedSession, "updatedAt">;
|
|
55
|
+
}>;
|
|
56
|
+
remove?: string[];
|
|
57
|
+
}, now?: number): void;
|
|
31
58
|
/** Forget a binding (session retire, Zen bucket re-roll). */
|
|
32
59
|
export declare function deletePersistedSession(family: string, key: string): void;
|
|
33
60
|
/** Drop expired bindings; returns the number pruned. */
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
import "./support/isolate-session-registry";
|
|
@@ -1 +1 @@
|
|
|
1
|
-
|
|
1
|
+
import "./support/isolate-session-registry";
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
import "./support/isolate-session-registry";
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -1 +1 @@
|
|
|
1
|
-
|
|
1
|
+
import "./support/isolate-session-registry";
|
|
@@ -1 +1 @@
|
|
|
1
|
-
|
|
1
|
+
import "./support/isolate-session-registry";
|
|
@@ -1 +1 @@
|
|
|
1
|
-
|
|
1
|
+
import "./support/isolate-session-registry";
|
|
@@ -1 +1 @@
|
|
|
1
|
-
|
|
1
|
+
import "./support/isolate-session-registry";
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { Transformer } from "../types/transformer";
|
|
2
|
+
import { type CodexBootstrapOptions } from "../utils/codex-bootstrap";
|
|
2
3
|
/**
|
|
3
4
|
* ChatGPT/Codex backend auth + Responses-wire constraints.
|
|
4
5
|
*
|
|
@@ -8,10 +9,26 @@ import { Transformer } from "../types/transformer";
|
|
|
8
9
|
* this transformer only stamps auth, Codex headers, `store: false`, and
|
|
9
10
|
* `stream: true`.
|
|
10
11
|
*/
|
|
12
|
+
export interface CodexTransformerOptions extends CodexBootstrapOptions {
|
|
13
|
+
streamBootstrapMaxFrames?: number;
|
|
14
|
+
streamBootstrapMaxBytes?: number;
|
|
15
|
+
streamBootstrapTimeoutMs?: number;
|
|
16
|
+
/**
|
|
17
|
+
* Hold pre-generation SSE frames uncommitted so an in-stream
|
|
18
|
+
* `server_is_overloaded` / quota rejection throws a fallback-eligible
|
|
19
|
+
* error *before* downstream headers commit, instead of reaching the
|
|
20
|
+
* client as a failed stream. Default false (opt-in).
|
|
21
|
+
*/
|
|
22
|
+
streamBootstrapBuffering?: boolean;
|
|
23
|
+
}
|
|
11
24
|
export declare class CodexTransformer implements Transformer {
|
|
25
|
+
static TransformerName: string;
|
|
12
26
|
name: string;
|
|
13
27
|
requestPhase: "headers";
|
|
14
28
|
logger?: any;
|
|
29
|
+
private readonly bootstrapOptions;
|
|
30
|
+
constructor(options?: CodexTransformerOptions);
|
|
31
|
+
private get bootstrapEnabled();
|
|
15
32
|
private streamIntent;
|
|
16
33
|
transformRequestIn(request: any, provider: any, context?: any): Promise<Record<string, any>>;
|
|
17
34
|
auth(request: any, provider: any): Promise<any>;
|
|
@@ -29,6 +46,7 @@ export declare class CodexTransformer implements Transformer {
|
|
|
29
46
|
req?: {
|
|
30
47
|
id?: string;
|
|
31
48
|
};
|
|
49
|
+
signal?: AbortSignal;
|
|
32
50
|
}): Promise<Response>;
|
|
33
51
|
private normalizeCodexTransport;
|
|
34
52
|
private ensureSseContentType;
|
package/dist/types/llm.d.ts
CHANGED
|
@@ -6,6 +6,7 @@ import type { ChatCompletionTool } from "openai/resources/chat/completions";
|
|
|
6
6
|
import type { Tool as AnthropicTool } from "@anthropic-ai/sdk/resources/messages";
|
|
7
7
|
import { Transformer } from "./transformer";
|
|
8
8
|
import type { ProviderTokenizerConfig } from "./tokenizer";
|
|
9
|
+
import type { RequestScopedErrorRule } from "../utils/request-scoped-errors";
|
|
9
10
|
export type TransformerConfigEntry = string | [string, Record<string, any>?];
|
|
10
11
|
export interface UrlCitation {
|
|
11
12
|
url: string;
|
|
@@ -259,6 +260,12 @@ export interface LLMProvider {
|
|
|
259
260
|
models: string[];
|
|
260
261
|
/** Optional GCP / Antigravity project id (provider-specific). */
|
|
261
262
|
project_id?: string;
|
|
263
|
+
/**
|
|
264
|
+
* Request-scoped error rules for this provider (CLIProxyAPI port).
|
|
265
|
+
* Evaluated before global rules; first match decides stop vs continue.
|
|
266
|
+
* The provider service stores rules only under this canonical key.
|
|
267
|
+
*/
|
|
268
|
+
request_scoped_errors?: RequestScopedErrorRule[];
|
|
262
269
|
transformer?: {
|
|
263
270
|
[key: string]: {
|
|
264
271
|
use?: Transformer[];
|
|
@@ -268,7 +275,18 @@ export interface LLMProvider {
|
|
|
268
275
|
passthrough?: any;
|
|
269
276
|
};
|
|
270
277
|
}
|
|
271
|
-
|
|
278
|
+
/**
|
|
279
|
+
* Accepted input spellings of the request-scoped error rule list. The
|
|
280
|
+
* provider service folds them into `request_scoped_errors`.
|
|
281
|
+
*/
|
|
282
|
+
export interface RequestScopedErrorsCarrier {
|
|
283
|
+
request_scoped_errors?: RequestScopedErrorRule[];
|
|
284
|
+
requestScopedErrors?: RequestScopedErrorRule[];
|
|
285
|
+
"request-scoped-errors"?: RequestScopedErrorRule[];
|
|
286
|
+
}
|
|
287
|
+
export type RegisterProviderRequest = LLMProvider & RequestScopedErrorsCarrier;
|
|
288
|
+
/** Provider update payload; any rule-list spelling replaces the stored rules. */
|
|
289
|
+
export type ProviderUpdate = Partial<LLMProvider> & RequestScopedErrorsCarrier;
|
|
272
290
|
export interface ModelRoute {
|
|
273
291
|
provider: string;
|
|
274
292
|
model: string;
|
|
@@ -279,7 +297,8 @@ export interface RequestRouteInfo {
|
|
|
279
297
|
originalModel: string;
|
|
280
298
|
targetModel: string;
|
|
281
299
|
}
|
|
282
|
-
|
|
300
|
+
/** Request-scoped error rules use any RequestScopedErrorsCarrier spelling. */
|
|
301
|
+
export interface ConfigProvider extends RequestScopedErrorsCarrier {
|
|
283
302
|
name: string;
|
|
284
303
|
api_base_url: string;
|
|
285
304
|
api_key: string;
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Codex stream bootstrap buffering (CLIProxyAPI `stream-bootstrap-buffering` port).
|
|
3
|
+
*
|
|
4
|
+
* The ChatGPT backend smuggles overload/quota rejections *inside* an HTTP 200
|
|
5
|
+
* SSE stream — right after the handshake events — instead of returning a
|
|
6
|
+
* retryable status on the wire. Once downstream headers are committed that
|
|
7
|
+
* failure can only be delivered to the client; holding the pre-generation
|
|
8
|
+
* bootstrap frames uncommitted keeps the window open for a transparent
|
|
9
|
+
* fallback to the next model.
|
|
10
|
+
*
|
|
11
|
+
* Held (never released on their own): handshake frames
|
|
12
|
+
* (`response.created`, `response.in_progress`, `response.queued`), `*.added`
|
|
13
|
+
* announcements, empty `output_text.delta` heartbeats, SSE comments, `event:`
|
|
14
|
+
* lines and blank separators. Anything else — text deltas, tool-call deltas,
|
|
15
|
+
* `*.done` / `*.completed` / `*.failed` / `error` frames, or an unparsable
|
|
16
|
+
* `data:` line — releases the buffer immediately. The hold is bounded by a
|
|
17
|
+
* frame budget and a byte budget, never by nothing.
|
|
18
|
+
*/
|
|
19
|
+
export interface CodexBootstrapOptions {
|
|
20
|
+
/** Max held `data:` lines before release. Default 48. */
|
|
21
|
+
maxFrames?: number;
|
|
22
|
+
/** Max held bytes before release. Default 1 MiB. */
|
|
23
|
+
maxBytes?: number;
|
|
24
|
+
/**
|
|
25
|
+
* Max hold time in ms before release. Default 0 (unlimited — the budget
|
|
26
|
+
* bounds what is held, not how long). Evaluated between reads; a peer that
|
|
27
|
+
* stops mid-line is still bounded by the request context, not this timer.
|
|
28
|
+
*/
|
|
29
|
+
timeoutMs?: number;
|
|
30
|
+
}
|
|
31
|
+
/**
|
|
32
|
+
* Capacity failure classes, as Codex CLI distinguishes them: the server is
|
|
33
|
+
* overloaded (503), the request rate is limited (429), or the account's quota
|
|
34
|
+
* or plan does not cover it (429). Each is worth another model; a request
|
|
35
|
+
* failure is not.
|
|
36
|
+
*/
|
|
37
|
+
export type CodexOverloadKind = "overload" | "rate_limit" | "quota";
|
|
38
|
+
export type CodexBootstrapRelease = "generated" | "budget" | "timeout" | "ended" | "overload";
|
|
39
|
+
export interface CodexBootstrapResult {
|
|
40
|
+
/** Replayable response: held bytes followed by the live remainder. */
|
|
41
|
+
response: Response;
|
|
42
|
+
overloaded: boolean;
|
|
43
|
+
overloadKind?: CodexOverloadKind;
|
|
44
|
+
/** Raw bootstrap text that carried the overload signal (truncated). */
|
|
45
|
+
overloadText?: string;
|
|
46
|
+
/** Structured error fields retained for internal request-scoped matching. */
|
|
47
|
+
overloadErrorText?: string;
|
|
48
|
+
/** Retry-After value (seconds or HTTP date) the failed event advised. */
|
|
49
|
+
overloadRetryAfter?: string;
|
|
50
|
+
heldBytes: number;
|
|
51
|
+
heldFrames: number;
|
|
52
|
+
releasedBy: CodexBootstrapRelease;
|
|
53
|
+
}
|
|
54
|
+
export declare const DEFAULT_BOOTSTRAP_MAX_FRAMES = 48;
|
|
55
|
+
export declare const DEFAULT_BOOTSTRAP_MAX_BYTES: number;
|
|
56
|
+
/** Capacity class of a structured Responses error; `message` is ignored. */
|
|
57
|
+
export declare function classifyCodexError(error: {
|
|
58
|
+
code?: unknown;
|
|
59
|
+
type?: unknown;
|
|
60
|
+
message?: unknown;
|
|
61
|
+
}): CodexOverloadKind | undefined;
|
|
62
|
+
/**
|
|
63
|
+
* Buffer the Codex SSE bootstrap. Never throws on upstream content: overload
|
|
64
|
+
* is reported on the result so the caller can raise a fallback-eligible error
|
|
65
|
+
* *before* downstream headers commit. The returned response replays held
|
|
66
|
+
* bytes first, then the untouched remainder of the original stream.
|
|
67
|
+
*/
|
|
68
|
+
export declare function bufferCodexBootstrapStream(response: Response, options?: CodexBootstrapOptions, signal?: AbortSignal): Promise<CodexBootstrapResult>;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OpenAI model metadata imported from Codex CLI's bundled catalog
|
|
3
|
+
* (codex-rs `models-manager/models.json`), so CCR sends what Codex sends:
|
|
4
|
+
*
|
|
5
|
+
* - reasoning effort: only the model's supported levels. `ultra` is a picker
|
|
6
|
+
* alias and never reaches the wire (`ModelInfo::resolve_reasoning_effort`).
|
|
7
|
+
* - Responses Lite: the request shape Codex uses for `use_responses_lite`
|
|
8
|
+
* models (`core/src/client.rs::build_responses_request`).
|
|
9
|
+
* - verbosity: only models with `support_verbosity`.
|
|
10
|
+
*
|
|
11
|
+
* Refresh the table when Codex adds or retires catalog models.
|
|
12
|
+
*/
|
|
13
|
+
import { type ReasoningSummaryLevel } from "./reasoning-effort";
|
|
14
|
+
import type { ThinkLevel } from "../types/llm";
|
|
15
|
+
export type CodexTextVerbosity = "low" | "medium" | "high";
|
|
16
|
+
export interface CodexModelSpec {
|
|
17
|
+
/** `supported_reasoning_levels`, lowest first. */
|
|
18
|
+
efforts: readonly ThinkLevel[];
|
|
19
|
+
/** `multi_agent_reasoning_effort`: what `ultra` resolves to, when set. */
|
|
20
|
+
ultraEffort?: ThinkLevel;
|
|
21
|
+
/** `use_responses_lite`. */
|
|
22
|
+
responsesLite: boolean;
|
|
23
|
+
/** `support_verbosity`. */
|
|
24
|
+
verbosity: boolean;
|
|
25
|
+
}
|
|
26
|
+
/**
|
|
27
|
+
* GPT-6 Luna slugs (`gpt-6-luna`, `gpt-6.1-luna`, `openai/gpt-6-luna`,
|
|
28
|
+
* `codex,gpt-6.1-luna`). Anchored so `gpt-6-sol` / `gpt-6-astra` /
|
|
29
|
+
* `gpt-5.6-luna` do not match.
|
|
30
|
+
*/
|
|
31
|
+
export declare function isGpt6LunaModel(model: unknown): boolean;
|
|
32
|
+
/**
|
|
33
|
+
* Catalog metadata for a model id, or undefined for models Codex does not
|
|
34
|
+
* list. GPT-6 slugs newer than the table get the family's shape (Luna tops
|
|
35
|
+
* out at `max`), so a new minor release keeps working before a refresh.
|
|
36
|
+
*/
|
|
37
|
+
export declare function codexModelSpec(model: unknown): CodexModelSpec | undefined;
|
|
38
|
+
/**
|
|
39
|
+
* The effort a catalog model is sent. `ultra` follows Codex: the model's
|
|
40
|
+
* multi-agent effort, else `max`, else the highest level below `ultra`. Any
|
|
41
|
+
* other unsupported level moves to the nearest end of the supported range
|
|
42
|
+
* (`none` / `minimal` → the lowest). Unknown tokens and unlisted models pass
|
|
43
|
+
* through unchanged.
|
|
44
|
+
*/
|
|
45
|
+
export declare function resolveCodexReasoningEffort(model: unknown, effort: ThinkLevel | undefined): ThinkLevel | undefined;
|
|
46
|
+
/**
|
|
47
|
+
* Apply `resolveCodexReasoningEffort` to a Responses/Unified request in
|
|
48
|
+
* place. Covers convert (`openai-responses`) and same-protocol wire-keep
|
|
49
|
+
* (`codex`). A disabled-reasoning request raised to a supported level is
|
|
50
|
+
* re-enabled, since the model cannot run without reasoning.
|
|
51
|
+
*/
|
|
52
|
+
export declare function applyCodexReasoningEffort(request: {
|
|
53
|
+
model?: unknown;
|
|
54
|
+
reasoning?: {
|
|
55
|
+
effort?: unknown;
|
|
56
|
+
enabled?: boolean;
|
|
57
|
+
} | null;
|
|
58
|
+
}): void;
|
|
59
|
+
/**
|
|
60
|
+
* `text.verbosity` implied by `REASONING_AUTO_SUMMARY`: detailed thinking
|
|
61
|
+
* pairs with verbose answers, concise with terse ones.
|
|
62
|
+
*/
|
|
63
|
+
export declare function verbosityForReasoningSummary(summary: ReasoningSummaryLevel | undefined): CodexTextVerbosity | undefined;
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
export declare function firstUserText(request: unknown): string;
|
|
3
3
|
export declare function isHarnessUserNoise(text: string): boolean;
|
|
4
4
|
export declare function userMessageTextParts(content: unknown): string[];
|
|
5
|
+
/** Substantive user turns, in order. Reminders and caveats are not identity. */
|
|
6
|
+
export declare function substantiveUserTexts(request: unknown): string[];
|
|
5
7
|
/**
|
|
6
8
|
* First user text that distinguishes a worker transcript.
|
|
7
9
|
* Shared reminder/caveat preambles are skipped so parallel Tasks do not collide.
|
|
@@ -9,6 +11,8 @@ export declare function userMessageTextParts(content: unknown): string[];
|
|
|
9
11
|
export declare function firstSubstantiveUserText(request: unknown): string;
|
|
10
12
|
/** Statusline / spinner polls — must not supersede or become a cache baseline. */
|
|
11
13
|
export declare function isStatuslinePollTurn(request: unknown): boolean;
|
|
14
|
+
/** Explicit worker-fork boundary, including inside inherited history. */
|
|
15
|
+
export declare function isForkOpeningText(text: string): boolean;
|
|
12
16
|
/**
|
|
13
17
|
* Nested/worker agent on any inbound protocol.
|
|
14
18
|
*
|
|
@@ -19,11 +19,47 @@ export declare function mapCallId(map: ResponsesCallIdMap, id: unknown, directio
|
|
|
19
19
|
* collisions resolve identically in both directions.
|
|
20
20
|
*/
|
|
21
21
|
export declare function sanitizeResponsesWireCallIds(body: any, callIdMap?: ResponsesCallIdMap): any;
|
|
22
|
+
/**
|
|
23
|
+
* Opt-in inbound compatibility for Codex multi-agent traffic (CLIProxyAPI
|
|
24
|
+
* `orphan-delegation-compatibility` / `optimize-multi-agent-v2` port).
|
|
25
|
+
*
|
|
26
|
+
* Both flags default to false and are read from top-level config
|
|
27
|
+
* (`orphanDelegationCompatibility` / `orphan_delegation_compatibility`,
|
|
28
|
+
* `optimizeMultiAgentV2` / `optimize_multi_agent_v2`). Orphan conversion
|
|
29
|
+
* additionally requires the `X-Openai-Subagent: collab_spawn` request header,
|
|
30
|
+
* so unrelated clients never silently change shape.
|
|
31
|
+
*/
|
|
32
|
+
export interface ResponsesInboundCompatOptions {
|
|
33
|
+
orphanDelegationCompatibility?: boolean;
|
|
34
|
+
optimizeMultiAgentV2?: boolean;
|
|
35
|
+
/** True when the collab_spawn subagent header is present on the request. */
|
|
36
|
+
subagentCollabSpawn?: boolean;
|
|
37
|
+
}
|
|
38
|
+
export declare function hasCollabSpawnSubagentHeader(headers: unknown): boolean;
|
|
39
|
+
export declare function resolveResponsesInboundCompat(headers: unknown, configService?: {
|
|
40
|
+
get(key: string): any;
|
|
41
|
+
}): ResponsesInboundCompatOptions;
|
|
42
|
+
/**
|
|
43
|
+
* Apply the opt-in multi-agent compatibility to the Responses client wire.
|
|
44
|
+
*
|
|
45
|
+
* Same-protocol wire keep (primary and fallback) sends `clientWireBody`
|
|
46
|
+
* upstream instead of the Unified projection, so the Unified-side conversion
|
|
47
|
+
* alone would still ship items the upstream cannot correlate or does not
|
|
48
|
+
* know. Under the same gates as responsesRequestToUnified:
|
|
49
|
+
* - orphan outputs (flag + `collab_spawn` header) become user message items,
|
|
50
|
+
* ordered exactly like the Unified projection;
|
|
51
|
+
* - role-less `agent_message` items (`optimizeMultiAgentV2`) become user
|
|
52
|
+
* message items with their content parts preserved.
|
|
53
|
+
* Role-bearing `agent_message` items and every other item stay byte-identical.
|
|
54
|
+
* Call only after normalization has validated the body. Returns the input
|
|
55
|
+
* body when nothing changes.
|
|
56
|
+
*/
|
|
57
|
+
export declare function repairResponsesWireMultiAgentCompat(body: any, compat: ResponsesInboundCompatOptions): any;
|
|
22
58
|
/**
|
|
23
59
|
* Client Responses wire → Unified (Chat Completions shape).
|
|
24
60
|
* Supports the Responses MVP subset; rejects CCR-unsupported stateful fields.
|
|
25
61
|
*/
|
|
26
|
-
export declare function responsesRequestToUnified(body: any, callIdMap?: ResponsesCallIdMap, customToolNames?: Set<string
|
|
62
|
+
export declare function responsesRequestToUnified(body: any, callIdMap?: ResponsesCallIdMap, customToolNames?: Set<string>, compat?: ResponsesInboundCompatOptions): UnifiedChatRequest;
|
|
27
63
|
/** Responses `include` is a string list; drop non-strings rather than invent values. */
|
|
28
64
|
export declare function normalizeResponsesInclude(include: unknown): string[] | undefined;
|
|
29
65
|
export declare function isResponsesReasoningItemId(value: unknown): value is string;
|
|
@@ -33,28 +33,11 @@ export declare function applyOpenAIChatReasoning(request: UnifiedChatRequest): U
|
|
|
33
33
|
/** Anthropic accepts low..max, while CCR/OpenAI may additionally emit minimal/ultra/none. */
|
|
34
34
|
export declare function toAnthropicReasoningEffort(effortValue: unknown): Exclude<ThinkLevel, "none" | "minimal" | "ultra"> | undefined;
|
|
35
35
|
/**
|
|
36
|
-
* GPT-6 family slugs (`gpt-6`, `gpt-6-astra`, `
|
|
37
|
-
* `codex,gpt-6-
|
|
36
|
+
* GPT-6 family slugs (`gpt-6`, `gpt-6-astra`, `gpt-6.1-sol`,
|
|
37
|
+
* `openai/gpt-6-astra`, `codex,gpt-6.1-sol`). Anchored so `gpt-60` /
|
|
38
|
+
* `gpt-5.6` do not match.
|
|
38
39
|
*/
|
|
39
40
|
export declare function isGpt6FamilyModel(model: unknown): boolean;
|
|
40
|
-
/** Astra/Sol reject `none` / `minimal`; OpenAI's migration floor is `low`. */
|
|
41
|
-
export declare function coerceGpt6ReasoningEffort(model: unknown, effort: ThinkLevel | undefined): ThinkLevel | undefined;
|
|
42
|
-
/**
|
|
43
|
-
* GPT-6 Luna slugs (`gpt-6-luna`, `openai/gpt-6-luna`, `codex,gpt-6-luna`).
|
|
44
|
-
* Anchored so `gpt-6-sol` / `gpt-6-astra` do not match.
|
|
45
|
-
*/
|
|
46
|
-
export declare function isGpt6LunaModel(model: unknown): boolean;
|
|
47
|
-
/**
|
|
48
|
-
* Remap unsupported GPT-6 efforts on a Responses/Unified request in place.
|
|
49
|
-
* Covers convert (`openai-responses`) and same-protocol wire-keep (`codex`).
|
|
50
|
-
*/
|
|
51
|
-
export declare function applyGpt6ReasoningEffortCoercion(request: {
|
|
52
|
-
model?: unknown;
|
|
53
|
-
reasoning?: {
|
|
54
|
-
effort?: unknown;
|
|
55
|
-
enabled?: boolean;
|
|
56
|
-
} | null;
|
|
57
|
-
}): void;
|
|
58
41
|
/**
|
|
59
42
|
* GPT-6 Astra rejects temperature / top_p / logprobs on Responses. Strip in
|
|
60
43
|
* place when the model is in the gpt-6 family.
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
export type RequestScopedErrorAction = "stop" | "stop-and-cooldown" | "continue" | "continue-and-cooldown";
|
|
2
|
+
export interface RequestScopedErrorRule {
|
|
3
|
+
status?: number;
|
|
4
|
+
match?: string[];
|
|
5
|
+
match_regex?: string[];
|
|
6
|
+
matchRegex?: string[];
|
|
7
|
+
"match-regex"?: string[];
|
|
8
|
+
action: RequestScopedErrorAction;
|
|
9
|
+
cooldown_seconds?: number;
|
|
10
|
+
cooldownSeconds?: number;
|
|
11
|
+
}
|
|
12
|
+
export interface NormalizedScopedErrorRule {
|
|
13
|
+
status?: number;
|
|
14
|
+
/** Lowercased substring patterns. */
|
|
15
|
+
match: string[];
|
|
16
|
+
matchRegex: RegExp[];
|
|
17
|
+
action: RequestScopedErrorAction;
|
|
18
|
+
cooldownSeconds: number;
|
|
19
|
+
}
|
|
20
|
+
/** A config rule rejected by validation; `index` is -1 for a non-array list. */
|
|
21
|
+
export interface ScopedErrorRuleIssue {
|
|
22
|
+
index: number;
|
|
23
|
+
reason: string;
|
|
24
|
+
}
|
|
25
|
+
/** Default cooldown for `*-and-cooldown` actions (matches legacy 60s transient). */
|
|
26
|
+
export declare const DEFAULT_SCOPED_ERROR_COOLDOWN_SECONDS = 60;
|
|
27
|
+
/** Accepted spellings of the rule-list key, canonical first. */
|
|
28
|
+
export declare const SCOPED_ERROR_RULE_KEYS: readonly ["request_scoped_errors", "requestScopedErrors", "request-scoped-errors"];
|
|
29
|
+
/**
|
|
30
|
+
* Compile raw config rules. An invalid rule (unknown key, bad action, status,
|
|
31
|
+
* pattern list, regex or cooldown) is dropped and reported via `onInvalid`;
|
|
32
|
+
* it is never widened into a broader match.
|
|
33
|
+
*/
|
|
34
|
+
export declare function normalizeRequestScopedErrorRules(input: unknown, onInvalid?: (issue: ScopedErrorRuleIssue) => void): NormalizedScopedErrorRule[];
|
|
35
|
+
/**
|
|
36
|
+
* Compiled rules for a raw config list, cached per list object so validation
|
|
37
|
+
* (and its `onInvalid` reports) runs once per config value, not per request.
|
|
38
|
+
*/
|
|
39
|
+
export declare function compiledScopedErrorRules(input: unknown, onInvalid?: (issue: ScopedErrorRuleIssue) => void): NormalizedScopedErrorRule[];
|
|
40
|
+
/** Read a rule list from a config carrier under any accepted key spelling. */
|
|
41
|
+
export declare function readScopedErrorRulesFromCarrier(carrier: unknown): unknown;
|
|
42
|
+
/** Top-level (global) rule list from a config service, any key spelling. */
|
|
43
|
+
export declare function readGlobalScopedErrorRules(configService: {
|
|
44
|
+
get(key: string): unknown;
|
|
45
|
+
}): unknown;
|
|
46
|
+
/** Upper bound (chars) on the raw text kept for rule matching. */
|
|
47
|
+
export declare const MAX_SCOPED_ERROR_CLASSIFICATION_CHARS = 16384;
|
|
48
|
+
/** Attach the raw upstream error text an error was built from. */
|
|
49
|
+
export declare function attachScopedErrorClassificationText(error: unknown, rawText: unknown): void;
|
|
50
|
+
/** Raw classification text previously attached to an error, if any. */
|
|
51
|
+
export declare function readScopedErrorClassificationText(error: unknown): string | undefined;
|
|
52
|
+
/** HTTP status for classification: explicit field first, then upstream snapshot. */
|
|
53
|
+
export declare function errorStatusForClassification(error: any): number | undefined;
|
|
54
|
+
/**
|
|
55
|
+
* Searchable body text. Prefers the raw upstream text a producer attached;
|
|
56
|
+
* without it, falls back to the message plus any captured upstream body. A
|
|
57
|
+
* specific error code is appended either way. The error `type` (always
|
|
58
|
+
* `api_error` for provider failures) is never included.
|
|
59
|
+
*/
|
|
60
|
+
export declare function errorTextForClassification(error: any): string;
|
|
61
|
+
/** First matching rule, or undefined when nothing matches. */
|
|
62
|
+
export declare function matchScopedErrorRule(error: any, rules: NormalizedScopedErrorRule[]): NormalizedScopedErrorRule | undefined;
|
|
63
|
+
export type ScopedErrorDecision = {
|
|
64
|
+
kind: "default";
|
|
65
|
+
} | {
|
|
66
|
+
kind: "stop" | "continue";
|
|
67
|
+
rule: NormalizedScopedErrorRule;
|
|
68
|
+
cooldown: boolean;
|
|
69
|
+
};
|
|
70
|
+
/**
|
|
71
|
+
* Classify an upstream error against provider rules first, then global rules.
|
|
72
|
+
* `stop` forces a terminal failure (no fallback); `continue` forces fallback
|
|
73
|
+
* eligibility even when the status alone would not qualify.
|
|
74
|
+
*/
|
|
75
|
+
export declare function decideScopedError(error: any, providerRules: NormalizedScopedErrorRule[], globalRules: NormalizedScopedErrorRule[]): ScopedErrorDecision;
|
|
76
|
+
/**
|
|
77
|
+
* Cooldown key. The model drops its `[1m]` context marker: the primary route
|
|
78
|
+
* strips it before dispatch while fallback entries keep it as configured,
|
|
79
|
+
* yet both name the same upstream model.
|
|
80
|
+
*/
|
|
81
|
+
export declare function scopedErrorCooldownKey(providerName: string, model: string): string;
|
|
82
|
+
export declare class ScopedErrorCooldownRegistry {
|
|
83
|
+
private readonly expiries;
|
|
84
|
+
put(providerName: string, model: string, cooldownSeconds?: number, now?: number): void;
|
|
85
|
+
isCooledDown(providerName: string, model: string, now?: number): boolean;
|
|
86
|
+
/** Drop every expired entry so keys that are never re-read cannot pile up. */
|
|
87
|
+
prune(now?: number): void;
|
|
88
|
+
/** Number of tracked entries, expired or not. */
|
|
89
|
+
get size(): number;
|
|
90
|
+
}
|
|
91
|
+
/** The cooldown registry owned by `scope`, created on first use. */
|
|
92
|
+
export declare function scopedErrorCooldownsFor(scope: object): ScopedErrorCooldownRegistry;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@caeliq/llms",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.73",
|
|
4
4
|
"description": "A universal LLM API transformation server",
|
|
5
5
|
"main": "dist/cjs/server.cjs",
|
|
6
6
|
"module": "dist/esm/server.mjs",
|
|
@@ -30,8 +30,8 @@
|
|
|
30
30
|
],
|
|
31
31
|
"dependencies": {
|
|
32
32
|
"@anthropic-ai/sdk": "^0.120.0",
|
|
33
|
-
"@caeliq/ccr-shared": "^2.1.
|
|
34
|
-
"@cursor/sdk": "^1.0.
|
|
33
|
+
"@caeliq/ccr-shared": "^2.1.15",
|
|
34
|
+
"@cursor/sdk": "^1.0.36",
|
|
35
35
|
"@fastify/cors": "^11.3.0",
|
|
36
36
|
"@fastify/rate-limit": "^11.2.0",
|
|
37
37
|
"@google/genai": "^2.18.0",
|