@bitkyc08/opencodex 2.57.0 → 2.58.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/gui/dist/assets/{index-Cz7CLdif.js → index-BbrHOIY0.js} +2 -2
- package/gui/dist/index.html +1 -1
- package/package.json +2 -2
- package/src/adapters/codebuddy/scaffold-guard.ts +5 -4
- package/src/adapters/command-code.ts +13 -4
- package/src/adapters/cursor/cursor-errors.ts +15 -0
- package/src/adapters/cursor/discovery.ts +65 -1
- package/src/adapters/cursor/live-transport.ts +5 -1
- package/src/adapters/cursor/protobuf-events.ts +110 -11
- package/src/adapters/cursor/protobuf-request.ts +19 -1
- package/src/adapters/cursor/text-toolcall.ts +230 -0
- package/src/adapters/cursor/thread-continuity.ts +67 -0
- package/src/adapters/cursor/types.ts +5 -0
- package/src/adapters/cursor.ts +55 -5
- package/src/adapters/google-http.ts +38 -13
- package/src/adapters/mimo-free.ts +32 -17
- package/src/adapters/ollama-native.ts +42 -8
- package/src/adapters/openai-responses/passthrough.ts +30 -4
- package/src/adapters/openai-responses/request-strips.ts +43 -0
- package/src/adapters/physical-send.ts +50 -0
- package/src/bridge/response-json.ts +1 -1
- package/src/bridge/sse.ts +1 -1
- package/src/claude/outbound.ts +14 -4
- package/src/cli/config-command.ts +35 -18
- package/src/cli/dispatch.ts +17 -4
- package/src/cli/index.ts +44 -2
- package/src/cli/system-command.ts +70 -1
- package/src/cli/uninstall-client-state.ts +12 -0
- package/src/codex/auth-context.ts +42 -8
- package/src/codex/desktop-switches.ts +145 -0
- package/src/codex/history-job.ts +5 -1
- package/src/codex/history-provider.ts +33 -4
- package/src/codex/history-worker.ts +14 -1
- package/src/codex/inject/remove.ts +145 -7
- package/src/codex/inject/restore.ts +204 -32
- package/src/codex/inject.ts +3 -7
- package/src/codex/loopback-target.ts +9 -0
- package/src/codex/native-profile-startup.ts +64 -20
- package/src/config/atomic-write.ts +83 -8
- package/src/config/schema/config-schema.ts +2 -0
- package/src/config/schema/leaf-validators.ts +1 -0
- package/src/generated/compatibility-version.json +132 -68
- package/src/lib/bounded-subprocess.ts +62 -10
- package/src/lib/windows-secret-acl.ts +151 -15
- package/src/lib/windows-user-principal.ts +5 -1
- package/src/providers/derive.ts +6 -0
- package/src/providers/model-discovery.ts +19 -7
- package/src/providers/registry/entries-core.ts +11 -0
- package/src/providers/registry/entries-extended.ts +50 -28
- package/src/providers/registry/model-seeds.ts +67 -17
- package/src/providers/registry/types.ts +9 -0
- package/src/responses/spill-store.ts +17 -0
- package/src/responses/state/body-policy.ts +25 -0
- package/src/responses/state/spill-queue.ts +8 -6
- package/src/responses/state.ts +3 -22
- package/src/router.ts +4 -0
- package/src/server/auth-cors.ts +1 -0
- package/src/server/index/websocket-handler.ts +48 -1
- package/src/server/management/config-routes.ts +27 -5
- package/src/server/models-capabilities.ts +24 -3
- package/src/server/responses/codex-ws-exchange.ts +65 -4
- package/src/server/responses/combo-stream-preflight.ts +68 -5
- package/src/server/responses/core-combo.ts +26 -0
- package/src/server/responses/core-options.ts +3 -0
- package/src/server/responses/fetch-helpers.ts +4 -1
- package/src/server/responses/native-injection-protocol.ts +42 -0
- package/src/server/responses/native-injection-replay.ts +105 -0
- package/src/server/responses/native-injection.ts +242 -0
- package/src/server/responses/native-response-control.ts +56 -0
- package/src/server/responses/native-response-json.ts +14 -0
- package/src/server/responses/native-response-output.ts +37 -0
- package/src/server/responses/native-steering-log.ts +44 -0
- package/src/server/responses/native-steering-policy.ts +49 -0
- package/src/server/responses/native-steering-replay.ts +126 -0
- package/src/server/responses/native-steering-settings.ts +76 -0
- package/src/server/responses/native-steering.ts +400 -0
- package/src/server/responses/native-tool-results.ts +130 -0
- package/src/server/responses/passthrough-delivery.ts +11 -0
- package/src/server/responses/passthrough-dispatch.ts +33 -1
- package/src/server/responses/request-prepare.ts +41 -0
- package/src/server/responses/ws-upstream.ts +21 -1
- package/src/server/stop-teardown.ts +8 -1
- package/src/server/ws-bridge.ts +16 -1
- package/src/service/cli.ts +13 -1
- package/src/types/config.ts +4 -0
- package/src/types/provider.ts +13 -0
|
@@ -437,20 +437,42 @@ export const deepseekReasoningMapFor = (modelId: string): Record<string, string>
|
|
|
437
437
|
// Coding Plan: the products use different exact allowlists and different base URLs.
|
|
438
438
|
// Evidence: https://help.aliyun.com/en/model-studio/token-plan-personal-overview
|
|
439
439
|
// https://help.aliyun.com/en/model-studio/token-plan-quickstart
|
|
440
|
+
// 260909 refresh, re-probed against the live gateway (both regions, both tiers):
|
|
441
|
+
// https://github.com/oliver-mee/alibaba-token-plan-wiki (machine-readable catalog).
|
|
442
|
+
// glm-5.3 / glm-5.3-flash removed from both Token Plan catalogs: they exist on Z.AI
|
|
443
|
+
// endpoints but the Token Plan gateway has never served either id (the 260826 seed
|
|
444
|
+
// propagated them across every GLM-carrying catalog; a selected row 404s).
|
|
445
|
+
// The Beijing preset keeps the Personal Edition subset; non-chat ids (audio/image/
|
|
446
|
+
// video families) stay out: they answer only on async endpoints openai-chat cannot
|
|
447
|
+
// reach. deepseek-v4-pro-0813 is callable but NOT listed by /models, which is the
|
|
448
|
+
// reason liveModels must stay false for this provider. deepseek-v4.1-flash is the
|
|
449
|
+
// 260910 DeepSeek rename row: listed on /models on both tiers and regions from 260915,
|
|
450
|
+
// hybrid thinking, vision via user message and tool result, json_object but not
|
|
451
|
+
// json_schema (see noJsonSchemaModels on the entries).
|
|
452
|
+
// Beijing serves the Personal Edition, so this is the Personal-tier roster probed
|
|
453
|
+
// 260909 (a strict subset of Team). deepseek-v4-pro-0813 stays out of the Beijing
|
|
454
|
+
// entry: its callability is only proven on Team keys, and no Personal key has been
|
|
455
|
+
// shown to reach it. The Beijing entry also shares the intl maps, so it carries a
|
|
456
|
+
// few orphan keys (kimi/glm-5/MiniMax rows); harmless, and one map beats two
|
|
457
|
+
// drifting ones.
|
|
440
458
|
export const ALIBABA_TOKEN_PLAN_MODELS = [
|
|
441
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
442
|
-
"
|
|
459
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
460
|
+
"deepseek-v4-pro", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "glm-5.2",
|
|
443
461
|
];
|
|
444
462
|
export const ALIBABA_TOKEN_PLAN_QWEN_MODELS = [
|
|
445
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
463
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-flash",
|
|
446
464
|
];
|
|
447
465
|
export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
448
466
|
"qwen3.8-max": ["text", "image"],
|
|
449
|
-
"qwen3.
|
|
467
|
+
"qwen3.8-flash": ["text", "image"],
|
|
468
|
+
"qwen3.7-max": ["text"],
|
|
450
469
|
"qwen3.7-plus": ["text", "image"],
|
|
451
470
|
"qwen3.6-flash": ["text", "image"],
|
|
452
|
-
"
|
|
453
|
-
"
|
|
471
|
+
"deepseek-v4-pro": ["text"],
|
|
472
|
+
"deepseek-v4-pro-0813": ["text"],
|
|
473
|
+
"deepseek-v4-flash-0731": ["text"],
|
|
474
|
+
// Vision probed on the plan gateway 260915 (user message and tool result, both 200).
|
|
475
|
+
"deepseek-v4.1-flash": ["text", "image"],
|
|
454
476
|
"glm-5.2": ["text"],
|
|
455
477
|
};
|
|
456
478
|
|
|
@@ -458,15 +480,18 @@ export const ALIBABA_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
|
458
480
|
// Multi-vendor lineup distinct from Beijing — includes DeepSeek V4 flash, Kimi K2.7, MiniMax.
|
|
459
481
|
// Evidence: https://www.alibabacloud.com/help/en/model-studio/token-plan-overview
|
|
460
482
|
// https://qwencloud.com/pricing/token-plan (qwen3.8 metadata)
|
|
483
|
+
// The Team Edition roster (Singapore), verified identical to the CN Team set on 260909.
|
|
484
|
+
// deepseek-v4-pro is restored: it remains callable on the plan gateway (probed 260909,
|
|
485
|
+
// listed on /models on both regions) after being dropped as "retired" upstream.
|
|
461
486
|
export const ALIBABA_INTL_TOKEN_PLAN_MODELS = [
|
|
462
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
463
|
-
"deepseek-v4-flash", "deepseek-v3.2",
|
|
487
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
488
|
+
"deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731", "deepseek-v4.1-flash", "deepseek-v3.2",
|
|
464
489
|
"kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5",
|
|
465
|
-
"glm-5.
|
|
490
|
+
"glm-5.2", "glm-5.1", "glm-5",
|
|
466
491
|
"MiniMax-M2.5",
|
|
467
492
|
];
|
|
468
493
|
export const ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS = [
|
|
469
|
-
"qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
494
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
470
495
|
];
|
|
471
496
|
|
|
472
497
|
// 260722 Tencent Cloud Coding Plan. The plan's model set is explicitly dynamic; these are the
|
|
@@ -543,24 +568,49 @@ export const VOLCENGINE_PLAN_TEXT_ONLY_MODELS = [
|
|
|
543
568
|
"doubao-seed-2.0-pro",
|
|
544
569
|
];
|
|
545
570
|
export const ALIBABA_INTL_TOKEN_PLAN_INPUT_MODALITIES: Record<string, string[]> = {
|
|
546
|
-
|
|
547
|
-
"qwen3.7-max": ["text", "image"],
|
|
548
|
-
"qwen3.7-plus": ["text", "image"],
|
|
571
|
+
...ALIBABA_TOKEN_PLAN_INPUT_MODALITIES,
|
|
549
572
|
"qwen3.6-plus": ["text", "image"],
|
|
550
|
-
"qwen3.6-flash": ["text", "image"],
|
|
551
573
|
"deepseek-v4-flash": ["text"],
|
|
552
574
|
"deepseek-v3.2": ["text"],
|
|
553
575
|
"kimi-k2.7-code": ["text", "image"],
|
|
554
576
|
"kimi-k2.6": ["text", "image"],
|
|
555
577
|
"kimi-k2.5": ["text", "image"],
|
|
556
|
-
"glm-5.3": ["text"],
|
|
557
|
-
"glm-5.3-flash": ["text", "image"],
|
|
558
|
-
"glm-5.2": ["text"],
|
|
559
578
|
"glm-5.1": ["text"],
|
|
560
579
|
"glm-5": ["text"],
|
|
561
580
|
"MiniMax-M2.5": ["text"],
|
|
562
581
|
};
|
|
563
582
|
|
|
583
|
+
// Shared Token Plan metadata (260909 gateway probes; output ceilings are max_tokens
|
|
584
|
+
// boundary probes: accept at N, reject at N+1).
|
|
585
|
+
export const QWEN38_FAMILY = ["qwen3.8-max", "qwen3.8-flash"];
|
|
586
|
+
export const ALIBABA_TOKEN_PLAN_CONTEXT_WINDOWS: Record<string, number> = {
|
|
587
|
+
"qwen3.8-max": 1_000_000, "qwen3.8-flash": 1_000_000, "qwen3.7-max": 1_000_000, "qwen3.7-plus": 1_000_000,
|
|
588
|
+
"qwen3.6-plus": 1_000_000, "qwen3.6-flash": 1_000_000,
|
|
589
|
+
"deepseek-v4-pro": 1_000_000, "deepseek-v4-pro-0813": 1_000_000, "deepseek-v4-flash": 1_000_000,
|
|
590
|
+
"deepseek-v4-flash-0731": 1_000_000, "deepseek-v4.1-flash": 1_000_000, "deepseek-v3.2": 131_072,
|
|
591
|
+
"kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 262_144,
|
|
592
|
+
"glm-5.2": 1_000_000, "glm-5.1": 202_752, "glm-5": 202_752,
|
|
593
|
+
"MiniMax-M2.5": 196_608,
|
|
594
|
+
};
|
|
595
|
+
export const ALIBABA_TOKEN_PLAN_MAX_OUTPUT_TOKENS: Record<string, number> = {
|
|
596
|
+
"qwen3.8-max": 131_072, "qwen3.8-flash": 131_072, "qwen3.7-max": 131_072, "qwen3.7-plus": 131_072,
|
|
597
|
+
"qwen3.6-plus": 65_536, "qwen3.6-flash": 65_536,
|
|
598
|
+
"deepseek-v4-pro": 393_216, "deepseek-v4-pro-0813": 393_216, "deepseek-v4-flash": 393_216,
|
|
599
|
+
"deepseek-v4-flash-0731": 393_216, "deepseek-v4.1-flash": 393_216, "deepseek-v3.2": 65_536,
|
|
600
|
+
"kimi-k2.7-code": 262_144, "kimi-k2.6": 262_144, "kimi-k2.5": 98_304,
|
|
601
|
+
"glm-5.2": 131_072, "glm-5.1": 128_000, "glm-5": 16_384,
|
|
602
|
+
"MiniMax-M2.5": 32_768,
|
|
603
|
+
};
|
|
604
|
+
export const ALIBABA_TOKEN_PLAN_NO_VISION = [
|
|
605
|
+
"qwen3.7-max", "deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash",
|
|
606
|
+
"deepseek-v4-flash-0731", "deepseek-v3.2", "glm-5.2", "glm-5.1", "glm-5", "MiniMax-M2.5",
|
|
607
|
+
];
|
|
608
|
+
export const ALIBABA_TOKEN_PLAN_PRESERVE_REASONING = [
|
|
609
|
+
"qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
|
|
610
|
+
"deepseek-v4-pro", "deepseek-v4-pro-0813", "deepseek-v4-flash", "deepseek-v4-flash-0731",
|
|
611
|
+
"deepseek-v4.1-flash", "glm-5.2",
|
|
612
|
+
];
|
|
613
|
+
|
|
564
614
|
// 260717 Kimi K3: the subscription endpoint uses one upstream id (`k3`) for both
|
|
565
615
|
// entitlement tiers. Bare `k3` advertises the Moderato 256K ceiling; the local `[1m]`
|
|
566
616
|
// alias advertises Allegretto's 1M ceiling and is stripped before the upstream request.
|
|
@@ -64,6 +64,10 @@ export interface ProviderModelDiscoveryFilter {
|
|
|
64
64
|
interface ProviderModelDiscoverySharedSpec {
|
|
65
65
|
/** Query parameters applied to the resolved discovery URL. */
|
|
66
66
|
query?: Readonly<Record<string, string>>;
|
|
67
|
+
/** Top-level response key containing model rows; defaults to `data`. */
|
|
68
|
+
envelopeKey?: string;
|
|
69
|
+
/** Model-row field containing the provider-native identifier; defaults to `id`. */
|
|
70
|
+
idField?: string;
|
|
67
71
|
/** Declarative eligibility rules evaluated against each untrusted model row. */
|
|
68
72
|
filter?: ProviderModelDiscoveryFilter;
|
|
69
73
|
/** Optional lower byte ceiling; the process-wide hard ceiling still wins. */
|
|
@@ -217,6 +221,11 @@ export interface ProviderRegistryEntry {
|
|
|
217
221
|
* to stay contiguous. This is seeded/backfilled like other fixed wire capabilities.
|
|
218
222
|
*/
|
|
219
223
|
requiresAdjacentResponsesToolResults?: boolean;
|
|
224
|
+
/**
|
|
225
|
+
* Responses upstream that also rejects a tool call with no matching output anywhere in the
|
|
226
|
+
* replayed input. Seeded/backfilled like other fixed wire capabilities.
|
|
227
|
+
*/
|
|
228
|
+
requiresPairedResponsesToolResults?: boolean;
|
|
220
229
|
/**
|
|
221
230
|
* When enabled, tool results that are present but empty are annotated on the wire.
|
|
222
231
|
* Seeded/backfilled like other fixed wire capabilities.
|
|
@@ -159,6 +159,23 @@ function spillNow(): number {
|
|
|
159
159
|
return spillNowOverride?.() ?? Date.now();
|
|
160
160
|
}
|
|
161
161
|
|
|
162
|
+
/**
|
|
163
|
+
* The spill deadline clock, shared with the shutdown drain in `state/spill-queue.ts`.
|
|
164
|
+
*
|
|
165
|
+
* Every deadline the shutdown path enforces has to read the same clock the work it
|
|
166
|
+
* budgets reads. When the drain measured its reserve on `Date.now()` while the ACL
|
|
167
|
+
* harden it was budgeting ran on this injected clock, a test could freeze the clock,
|
|
168
|
+
* believe it had removed wall time from the case, and still lose an 80 ms reserve to
|
|
169
|
+
* real elapsed time on a loaded runner — which is what turned
|
|
170
|
+
* `shutdown fallback prices the job-owned superseded generation before publishing`
|
|
171
|
+
* red on macOS 2/2 in run 35137850114 while the assertion it was written for never ran.
|
|
172
|
+
*
|
|
173
|
+
* Production is unchanged: with no override installed this is `Date.now()`.
|
|
174
|
+
*/
|
|
175
|
+
export function responseSpillNow(): number {
|
|
176
|
+
return spillNow();
|
|
177
|
+
}
|
|
178
|
+
|
|
162
179
|
function record(event: "write" | "fsync" | "close" | "harden" | "publish" | "dir-fsync" | "stub-swap"): void {
|
|
163
180
|
spillIoForTest?.record?.(event);
|
|
164
181
|
}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Request bodies that must never enter the continuation cache.
|
|
3
|
+
*
|
|
4
|
+
* The cache is persisted to `responses-state.json`, so anything recorded here reaches disk.
|
|
5
|
+
* Encrypted-agent-task recovery decrypts task text into the request body and promises
|
|
6
|
+
* in-memory, TTL-bounded retention; recording that body would put the plaintext on disk with
|
|
7
|
+
* no TTL and break the promise.
|
|
8
|
+
*
|
|
9
|
+
* A WeakSet rather than a body field on purpose: `_rawBody` is serialized verbatim by the
|
|
10
|
+
* native passthrough, so any marker written into the body itself would be sent upstream.
|
|
11
|
+
* Marking is enforced once here rather than at each call site, because every recording path
|
|
12
|
+
* (streaming, non-streaming, passthrough, forced) funnels through `rememberResponseState` —
|
|
13
|
+
* a new call site cannot reintroduce the leak by forgetting a guard.
|
|
14
|
+
*/
|
|
15
|
+
const nonPersistableBodies = new WeakSet<object>();
|
|
16
|
+
|
|
17
|
+
/** Bar this exact request body from the continuation cache, and therefore from disk. */
|
|
18
|
+
export function markBodyNonPersistable(body: unknown): void {
|
|
19
|
+
if (body && typeof body === "object") nonPersistableBodies.add(body as object);
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** Test the body's in-memory persistence restriction without adding a wire marker. */
|
|
23
|
+
export function isBodyNonPersistable(body: unknown): boolean {
|
|
24
|
+
return !!body && typeof body === "object" && nonPersistableBodies.has(body);
|
|
25
|
+
}
|
|
@@ -7,6 +7,7 @@ import {
|
|
|
7
7
|
MAX_RESPONSE_SPILL_PAYLOAD_BYTES,
|
|
8
8
|
prospectiveResponseSpillBytes,
|
|
9
9
|
responseSpillPayloadCap,
|
|
10
|
+
responseSpillNow,
|
|
10
11
|
type ResponseSpillPublicationControl,
|
|
11
12
|
type ResponseSpillRef,
|
|
12
13
|
writeResponseSpillDurably,
|
|
@@ -377,7 +378,7 @@ function responseSpillShutdownBudget(): { totalMs: number; fallbackReserveMs: nu
|
|
|
377
378
|
}
|
|
378
379
|
|
|
379
380
|
function awaitResponseSpillTailUntil(observed: Promise<void>, deadline: number): Promise<boolean> {
|
|
380
|
-
const remaining = deadline -
|
|
381
|
+
const remaining = deadline - responseSpillNow();
|
|
381
382
|
if (remaining <= 0) return Promise.resolve(false);
|
|
382
383
|
return new Promise(resolve => {
|
|
383
384
|
let finished = false;
|
|
@@ -554,12 +555,13 @@ function terminalizeExhaustedShutdownFallback(
|
|
|
554
555
|
}
|
|
555
556
|
|
|
556
557
|
function fallbackPendingResponseSpills(reserveMs: number): Error[] {
|
|
557
|
-
|
|
558
|
+
// Same clock as the harden work this reserve is budgeting — see `responseSpillNow`.
|
|
559
|
+
const deadline = responseSpillNow() + reserveMs;
|
|
558
560
|
const failures: Error[] = [];
|
|
559
561
|
for (;;) {
|
|
560
562
|
const pending = pendingShutdownFallbackCandidates();
|
|
561
563
|
if (pending.length === 0) return failures;
|
|
562
|
-
if (
|
|
564
|
+
if (responseSpillNow() >= deadline) {
|
|
563
565
|
terminalizeExhaustedShutdownFallback(pending, failures);
|
|
564
566
|
return failures;
|
|
565
567
|
}
|
|
@@ -569,7 +571,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
|
|
|
569
571
|
for (let index = 0; index < pending.length; index += 1) {
|
|
570
572
|
const { job, candidate } = pending[index]!;
|
|
571
573
|
if (requireStore().currentEntry(job.id) !== candidate) continue;
|
|
572
|
-
const remaining = deadline -
|
|
574
|
+
const remaining = deadline - responseSpillNow();
|
|
573
575
|
if (remaining <= 0) {
|
|
574
576
|
reserveExhausted = true;
|
|
575
577
|
for (const exhausted of pending.slice(index)) {
|
|
@@ -587,7 +589,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
|
|
|
587
589
|
requireStore().recomputeOldestResident();
|
|
588
590
|
requireStore().pruneResponses();
|
|
589
591
|
enforceAppOwnedMemoryBudget();
|
|
590
|
-
if (reserveExhausted ||
|
|
592
|
+
if (reserveExhausted || responseSpillNow() >= deadline) {
|
|
591
593
|
terminalizeExhaustedShutdownFallback(pendingShutdownFallbackCandidates(), failures);
|
|
592
594
|
return failures;
|
|
593
595
|
}
|
|
@@ -597,7 +599,7 @@ function fallbackPendingResponseSpills(reserveMs: number): Error[] {
|
|
|
597
599
|
export async function drainResponseSpillPublications(): Promise<void> {
|
|
598
600
|
const budget = responseSpillShutdownBudget();
|
|
599
601
|
const fallbackReserveMs = Math.min(budget.totalMs, Math.max(1, budget.fallbackReserveMs));
|
|
600
|
-
const drainDeadline =
|
|
602
|
+
const drainDeadline = responseSpillNow() + Math.max(0, budget.totalMs - fallbackReserveMs);
|
|
601
603
|
|
|
602
604
|
for (;;) {
|
|
603
605
|
if (pendingResponseSpills.size === 0) return;
|
package/src/responses/state.ts
CHANGED
|
@@ -23,6 +23,8 @@ import type { ResponseSpillWriteFailureCode, ResponseSpillWriteStatus, ResponseS
|
|
|
23
23
|
export { responseAdmissionCountersForTests } from "./state/spill-failure";
|
|
24
24
|
import { admissionCounters, noteSpillWriteFailure, noteSpillWriteSuccess, spillCounters, spillWriteHealth } from "./state/spill-failure";
|
|
25
25
|
import { loadSnapshotEntry } from "./state/snapshot-codec";
|
|
26
|
+
import { isBodyNonPersistable } from "./state/body-policy";
|
|
27
|
+
export { isBodyNonPersistable, markBodyNonPersistable } from "./state/body-policy";
|
|
26
28
|
export { flushPendingResponseSpillsForTests, awaitResponseSpillPublicationTailForTests, pendingResponseSpillMetricsForTests, setResponseSpillShutdownBudgetForTests, setResponseSpillAsyncAclAttemptBudgetForTests, setResponseSpillShutdownTerminalizationPassLimitForTests } from "./state/spill-queue";
|
|
27
29
|
import {
|
|
28
30
|
bindSpillQueueStore,
|
|
@@ -1235,27 +1237,6 @@ export function responseStateMetrics(): ResponseStateMetrics {
|
|
|
1235
1237
|
* Cache completed output and max_output_tokens partial output for previous_response_id replay.
|
|
1236
1238
|
* Content-filtered incomplete and failed output are not authoritative replay history.
|
|
1237
1239
|
*/
|
|
1238
|
-
/**
|
|
1239
|
-
* Request bodies that must never enter the continuation cache.
|
|
1240
|
-
*
|
|
1241
|
-
* The cache is persisted to `responses-state.json`, so anything recorded here reaches disk.
|
|
1242
|
-
* Encrypted-agent-task recovery decrypts task text into the request body and promises
|
|
1243
|
-
* in-memory, TTL-bounded retention; recording that body would put the plaintext on disk with
|
|
1244
|
-
* no TTL and break the promise.
|
|
1245
|
-
*
|
|
1246
|
-
* A WeakSet rather than a body field on purpose: `_rawBody` is serialized verbatim by the
|
|
1247
|
-
* native passthrough, so any marker written into the body itself would be sent upstream.
|
|
1248
|
-
* Marking is enforced once here rather than at each call site, because every recording path
|
|
1249
|
-
* (streaming, non-streaming, passthrough, forced) funnels through `rememberResponseState` —
|
|
1250
|
-
* a new call site cannot reintroduce the leak by forgetting a guard.
|
|
1251
|
-
*/
|
|
1252
|
-
const nonPersistableBodies = new WeakSet<object>();
|
|
1253
|
-
|
|
1254
|
-
/** Bar this exact request body from the continuation cache, and therefore from disk. */
|
|
1255
|
-
export function markBodyNonPersistable(body: unknown): void {
|
|
1256
|
-
if (body && typeof body === "object") nonPersistableBodies.add(body as object);
|
|
1257
|
-
}
|
|
1258
|
-
|
|
1259
1240
|
export function rememberResponseState(
|
|
1260
1241
|
requestBody: unknown,
|
|
1261
1242
|
response: { id?: unknown; output?: unknown; status?: unknown; incomplete_details?: unknown },
|
|
@@ -1264,7 +1245,7 @@ export function rememberResponseState(
|
|
|
1264
1245
|
): void {
|
|
1265
1246
|
if (!requestBody || typeof requestBody !== "object" || Array.isArray(requestBody)) return;
|
|
1266
1247
|
const request = requestBody as Record<string, unknown>;
|
|
1267
|
-
if (
|
|
1248
|
+
if (isBodyNonPersistable(request)) return;
|
|
1268
1249
|
// `force` bypasses only the store:false skip: Codex sends `store:false` on every non-Azure
|
|
1269
1250
|
// HTTP request (and WS inherits it), yet its WS turns still chain with previous_response_id.
|
|
1270
1251
|
// The passthrough branch records with force so those chains can be expanded locally; the
|
package/src/router.ts
CHANGED
|
@@ -384,6 +384,10 @@ export function routedProviderConfig(providerName: string, provider: OcxProvider
|
|
|
384
384
|
&& registryEntry.requiresAdjacentResponsesToolResults !== undefined
|
|
385
385
|
? { requiresAdjacentResponsesToolResults: registryEntry.requiresAdjacentResponsesToolResults }
|
|
386
386
|
: {}),
|
|
387
|
+
...(provider.requiresPairedResponsesToolResults === undefined
|
|
388
|
+
&& registryEntry.requiresPairedResponsesToolResults !== undefined
|
|
389
|
+
? { requiresPairedResponsesToolResults: registryEntry.requiresPairedResponsesToolResults }
|
|
390
|
+
: {}),
|
|
387
391
|
...(provider.annotateEmptyToolOutputs === undefined
|
|
388
392
|
&& registryEntry.annotateEmptyToolOutputs !== undefined
|
|
389
393
|
? { annotateEmptyToolOutputs: registryEntry.annotateEmptyToolOutputs }
|
package/src/server/auth-cors.ts
CHANGED
|
@@ -913,6 +913,7 @@ const PROVIDER_CONFIG_FIELD_POLICY = {
|
|
|
913
913
|
commandCodeVersion: "editor",
|
|
914
914
|
statelessResponses: "editor",
|
|
915
915
|
requiresAdjacentResponsesToolResults: "editor",
|
|
916
|
+
requiresPairedResponsesToolResults: "editor",
|
|
916
917
|
annotateEmptyToolOutputs: "editor",
|
|
917
918
|
supportsServiceTier: "editor",
|
|
918
919
|
modelSupportsServiceTier: "editor",
|
|
@@ -1,3 +1,7 @@
|
|
|
1
|
+
import { nativeSteeringUnavailableReason, nativeResponseControlMode, type NativeResponseControl } from "../responses/native-response-control";
|
|
2
|
+
import { NativeInjectionChannel } from "../responses/native-injection";
|
|
3
|
+
import { NativeSteeringChannel, NativeSteeringError } from "../responses/native-steering";
|
|
4
|
+
import { createNativeSteeringLogObserver } from "../responses/native-steering-log";
|
|
1
5
|
import type { Server, ServerWebSocket } from "bun";
|
|
2
6
|
import {
|
|
3
7
|
LIVE_SIDEBAND_UPSTREAM_OPEN_TIMEOUT_MS,
|
|
@@ -188,11 +192,47 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
|
|
|
188
192
|
} catch {
|
|
189
193
|
return; // text-only contract; ignore unparseable frames
|
|
190
194
|
}
|
|
195
|
+
if (frame.type === "response.inject" || frame.type === "response.steer" || (frame.type === "response.create" && ws.data.nativeControl)) {
|
|
196
|
+
try {
|
|
197
|
+
if (frame.type === "response.inject") {
|
|
198
|
+
if (!ws.data.nativeControl?.inject) throw new NativeSteeringError("injection_not_supported", "Native injection is disabled or unavailable on this route.");
|
|
199
|
+
ws.data.nativeControl.inject(frame);
|
|
200
|
+
return;
|
|
201
|
+
}
|
|
202
|
+
if (frame.type === "response.steer") {
|
|
203
|
+
if (!ws.data.nativeControl) throw new NativeSteeringError("steering_not_supported", ws.data.nativeSteeringUnavailable ?? "Native steering transport is unavailable; the route may be unsupported or using HTTP fallback.");
|
|
204
|
+
ws.data.nativeControl.steer(frame);
|
|
205
|
+
return;
|
|
206
|
+
}
|
|
207
|
+
if (ws.data.nativeControl?.continue(frame)) return;
|
|
208
|
+
} catch (error) {
|
|
209
|
+
sendJsonFrame(ws, buildWsErrorFrame(400, {
|
|
210
|
+
type: "invalid_request_error",
|
|
211
|
+
code: error instanceof NativeSteeringError ? error.code : "native_steering_error",
|
|
212
|
+
message: error instanceof NativeSteeringError ? error.message : "Native steering transport failed; delivery may be unknown. Do not automatically replay input.",
|
|
213
|
+
}));
|
|
214
|
+
return;
|
|
215
|
+
}
|
|
216
|
+
}
|
|
191
217
|
if (frame.type === "response.processed") return; // ack — no-op
|
|
192
218
|
if (frame.type !== "response.create") return;
|
|
193
219
|
markActivity("ws response.create");
|
|
194
220
|
|
|
221
|
+
let nativeControl: NativeResponseControl | undefined;
|
|
222
|
+
try {
|
|
223
|
+
const idleMs = typeof config.stallTimeoutSec === "number" && Number.isFinite(config.stallTimeoutSec)
|
|
224
|
+
? Math.max(1, config.stallTimeoutSec) * 1000 : 300_000;
|
|
225
|
+
const mode = nativeResponseControlMode(frame, config);
|
|
226
|
+
nativeControl = mode === "injection" ? new NativeInjectionChannel(frame, idleMs)
|
|
227
|
+
: mode === "steering" ? new NativeSteeringChannel(frame, idleMs) : undefined;
|
|
228
|
+
} catch {
|
|
229
|
+
sendJsonFrame(ws, buildWsErrorFrame(400, { type: "invalid_request_error", message: "Invalid native steering request settings" }));
|
|
230
|
+
return;
|
|
231
|
+
}
|
|
195
232
|
ws.data.cancel?.();
|
|
233
|
+
// A superseded turn must not keep ownership during warmup or refusal.
|
|
234
|
+
ws.data.nativeControl = undefined;
|
|
235
|
+
ws.data.nativeSteeringUnavailable = nativeSteeringUnavailableReason(frame, config.codexNativeSteering);
|
|
196
236
|
const turnId = (ws.data.turnId ?? 0) + 1;
|
|
197
237
|
ws.data.turnId = turnId;
|
|
198
238
|
const isCurrent = () => ws.data.turnId === turnId;
|
|
@@ -227,6 +267,8 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
|
|
|
227
267
|
return;
|
|
228
268
|
}
|
|
229
269
|
|
|
270
|
+
// Only a genuinely admitted turn may receive steering or continuations.
|
|
271
|
+
ws.data.nativeControl = nativeControl;
|
|
230
272
|
const payload: Record<string, unknown> = { ...frame };
|
|
231
273
|
delete payload.type;
|
|
232
274
|
turnAdmissionLease.bindAbortController(turnAbort);
|
|
@@ -267,6 +309,7 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
|
|
|
267
309
|
...(wsAdmission ? { admission: wsAdmission } : {}),
|
|
268
310
|
forceEmptyResponseId: true,
|
|
269
311
|
inboundTransport: "websocket",
|
|
312
|
+
nativeControl,
|
|
270
313
|
abortSignal: turnAbort.signal,
|
|
271
314
|
turnAdmissionLease,
|
|
272
315
|
onFirstOutput: () => recordFirstOutput(logCtx, start),
|
|
@@ -277,7 +320,10 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
|
|
|
277
320
|
},
|
|
278
321
|
});
|
|
279
322
|
await sendResponseToWebSocket(ws, response, isCurrent, {
|
|
280
|
-
|
|
323
|
+
untilEof: nativeControl?.relayActive === true,
|
|
324
|
+
onSsePayload: nativeControl?.relayActive
|
|
325
|
+
? createNativeSteeringLogObserver(logCtx, () => recordFirstOutput(logCtx, start))
|
|
326
|
+
: payload => inspectResponseLogSsePayload(logCtx, payload),
|
|
281
327
|
onTerminal: status => {
|
|
282
328
|
terminalRecorder?.(status, logCtx.terminalHttpStatus);
|
|
283
329
|
finalizeLog(httpStatusForRequestLogTerminal(status, logCtx), {
|
|
@@ -313,6 +359,7 @@ export function createWebsocketHandler(ctx: ServeOptionsContext) {
|
|
|
313
359
|
}
|
|
314
360
|
} finally {
|
|
315
361
|
turnAdmissionLease.release();
|
|
362
|
+
if (ws.data.nativeControl === nativeControl) ws.data.nativeControl = undefined;
|
|
316
363
|
if (!logged && turnAbort.signal.aborted) finalizeLog(499);
|
|
317
364
|
if (ws.data.cancel === cancelTurn) ws.data.cancel = undefined;
|
|
318
365
|
}
|
|
@@ -3,6 +3,11 @@ import { randomUUID } from "node:crypto";
|
|
|
3
3
|
import { readFileSync } from "node:fs";
|
|
4
4
|
import type { CatalogModel } from "../../codex/catalog";
|
|
5
5
|
import { catalogModelSlug, invalidateCodexModelsCache, nativeContextLimits, nativeModelRows, uniqueCatalogModelsForPublicList } from "../../codex/catalog";
|
|
6
|
+
import {
|
|
7
|
+
applyCodexDesktopSwitches,
|
|
8
|
+
describeCodexDesktopSwitches,
|
|
9
|
+
type CodexDesktopSwitchApply,
|
|
10
|
+
} from "../../codex/desktop-switches";
|
|
6
11
|
import {
|
|
7
12
|
DEFAULT_SUBAGENT_MODELS,
|
|
8
13
|
codexAutoStartEnabled,
|
|
@@ -328,6 +333,11 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise<Respon
|
|
|
328
333
|
codexDesktopAuthless: config.codexDesktopAuthless === true,
|
|
329
334
|
// Absent keeps Design B remote compaction; true selects the dedicated provider identity.
|
|
330
335
|
codexClientCompaction: config.codexClientCompaction === true,
|
|
336
|
+
codexDesktopSwitches: describeCodexDesktopSwitches(config, {
|
|
337
|
+
applied: false,
|
|
338
|
+
reason: "not_requested",
|
|
339
|
+
retryable: false,
|
|
340
|
+
}),
|
|
331
341
|
startupHealth: await readStartupHealth(config),
|
|
332
342
|
codexRuntime: {
|
|
333
343
|
path: displayCodexRuntimePath(resolved.runtime.command),
|
|
@@ -597,15 +607,26 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise<Respon
|
|
|
597
607
|
configureAppOwnedMemoryBudget(resolveAppOwnedMemoryBudgetBytes(body.appOwnedMemoryBudgetMb));
|
|
598
608
|
enforceAppOwnedMemoryBudget();
|
|
599
609
|
}
|
|
600
|
-
// Both Desktop compatibility switches change the injected config.toml shape, so converge now
|
|
601
|
-
// rather than waiting for the next start; the injector re-reads config and rewrites the form.
|
|
602
610
|
const authlessIsEnabled = config.codexDesktopAuthless === true;
|
|
603
611
|
const clientCompactionIsEnabled = config.codexClientCompaction === true;
|
|
604
|
-
const
|
|
605
|
-
||
|
|
606
|
-
|
|
612
|
+
const desktopSwitchesChanged = authlessWasEnabled !== authlessIsEnabled
|
|
613
|
+
|| clientCompactionWasEnabled !== clientCompactionIsEnabled;
|
|
614
|
+
// Catalog convergence is not config injection, and the comment that used to sit here said
|
|
615
|
+
// it was. `convergeCodexCatalog` rejects any scope but `catalog` and never reaches the
|
|
616
|
+
// injector, which is why flipping either switch left `config.toml` in its old shape until
|
|
617
|
+
// a separate `ocx sync` (#4809). Both halves are needed when a Desktop switch changes; a
|
|
618
|
+
// picker-only update still refreshes just the catalog.
|
|
619
|
+
const catalogRefresh = pickerWasEnabled !== pickerIsEnabled || desktopSwitchesChanged
|
|
607
620
|
? await convergeCodexCatalog()
|
|
608
621
|
: undefined;
|
|
622
|
+
// Injection second, matching `syncModelsToCodex`: the injected `model_catalog_json` should
|
|
623
|
+
// point at a catalog that has already settled. And it runs here rather than inside the save
|
|
624
|
+
// because coordinated Codex writes acquire the Codex write lock N before the config mutation
|
|
625
|
+
// lock C — awaiting N while still holding C would invert that order.
|
|
626
|
+
const desktopSwitchApply: CodexDesktopSwitchApply = desktopSwitchesChanged
|
|
627
|
+
? await applyCodexDesktopSwitches(config)
|
|
628
|
+
: { applied: false, reason: "not_requested", retryable: false };
|
|
629
|
+
const codexDesktopSwitches = describeCodexDesktopSwitches(config, desktopSwitchApply);
|
|
609
630
|
const catalogRefreshPending = catalogRefresh
|
|
610
631
|
? catalogRefreshIsPending(catalogRefresh)
|
|
611
632
|
: false;
|
|
@@ -622,6 +643,7 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise<Respon
|
|
|
622
643
|
catalogRefreshPending,
|
|
623
644
|
codexDesktopAuthless: authlessIsEnabled,
|
|
624
645
|
codexClientCompaction: clientCompactionIsEnabled,
|
|
646
|
+
codexDesktopSwitches,
|
|
625
647
|
codexMainAccountHardLock: config.codexMainAccountHardLock === true,
|
|
626
648
|
mainAccountHardLock: getMainAccountHardLockStatus(config),
|
|
627
649
|
startupHealth: await readStartupHealth(config),
|
|
@@ -10,6 +10,11 @@ import type { CursorEffortTable } from "../integrations/cursor-effort-table";
|
|
|
10
10
|
* Completions, Responses and Anthropic Messages, streams, and accepts tool calls, so those are
|
|
11
11
|
* constants; context length and vision come from catalog data when known and are omitted
|
|
12
12
|
* otherwise, matching Cursor's optional-field schema.
|
|
13
|
+
*
|
|
14
|
+
* Top-level capacity metrics (`context_window`, `context_length`, `max_output_tokens`) are
|
|
15
|
+
* mirrored directly on each model row for external client discovery (e.g. pi-ai, DSH,
|
|
16
|
+
* LibreChat) that inspects flat properties rather than Cursor's nested `capabilities.*` shape.
|
|
17
|
+
* A row that gains a nested capacity value must gain the top-level mirror in the same change.
|
|
13
18
|
*/
|
|
14
19
|
|
|
15
20
|
/**
|
|
@@ -132,6 +137,19 @@ export interface ModelCapabilityFields {
|
|
|
132
137
|
supports_vision?: boolean;
|
|
133
138
|
reasoning_effort?: string[];
|
|
134
139
|
};
|
|
140
|
+
/**
|
|
141
|
+
* Mirrored top-level context window for external/legacy client discovery (e.g. pi-ai, DSH)
|
|
142
|
+
* that reads top-level context_window / context_length instead of nested capabilities.
|
|
143
|
+
*/
|
|
144
|
+
context_window?: number;
|
|
145
|
+
/**
|
|
146
|
+
* Top-level context length alias matching capabilities.context_length for clients expecting context_length.
|
|
147
|
+
*/
|
|
148
|
+
context_length?: number;
|
|
149
|
+
/**
|
|
150
|
+
* Mirrored top-level max output token limit for external/legacy client discovery.
|
|
151
|
+
*/
|
|
152
|
+
max_output_tokens?: number;
|
|
135
153
|
/**
|
|
136
154
|
* Cursor reads the long-context threshold from `pricing.overrides[].min_prompt_tokens`. That
|
|
137
155
|
* key sits outside its validated capability schema, so it is the one place a threshold can
|
|
@@ -153,6 +171,7 @@ export function modelCapabilityFields(input: ModelCapabilityInput): ModelCapabil
|
|
|
153
171
|
const longContextLength = positiveInt(input.longContextWindow);
|
|
154
172
|
const maxOutputTokens = positiveInt(input.maxOutputTokens);
|
|
155
173
|
const hasLongTier = contextLength !== undefined && longContextLength !== undefined && longContextLength > contextLength;
|
|
174
|
+
const effectiveContextLength = hasLongTier ? longContextLength : contextLength;
|
|
156
175
|
const modalities = Array.isArray(input.inputModalities)
|
|
157
176
|
? input.inputModalities.filter(modality => typeof modality === "string" && modality.length > 0)
|
|
158
177
|
: undefined;
|
|
@@ -160,9 +179,7 @@ export function modelCapabilityFields(input: ModelCapabilityInput): ModelCapabil
|
|
|
160
179
|
return {
|
|
161
180
|
api_types: [...OPENCODEX_MODEL_API_TYPES],
|
|
162
181
|
capabilities: {
|
|
163
|
-
...(
|
|
164
|
-
? { context_length: longContextLength }
|
|
165
|
-
: contextLength !== undefined ? { context_length: contextLength } : {}),
|
|
182
|
+
...(effectiveContextLength !== undefined ? { context_length: effectiveContextLength } : {}),
|
|
166
183
|
...(maxOutputTokens !== undefined ? { max_output_tokens: maxOutputTokens } : {}),
|
|
167
184
|
// Once a gateway advertises api_types, Cursor keeps only rows whose output_modalities
|
|
168
185
|
// include "text"; omitting the key drops the row from the extended catalog.
|
|
@@ -174,6 +191,10 @@ export function modelCapabilityFields(input: ModelCapabilityInput): ModelCapabil
|
|
|
174
191
|
...(supportsVision !== undefined ? { supports_vision: supportsVision } : {}),
|
|
175
192
|
...(efforts.length > 0 ? { reasoning_effort: [...efforts] } : {}),
|
|
176
193
|
},
|
|
194
|
+
...(effectiveContextLength !== undefined
|
|
195
|
+
? { context_window: effectiveContextLength, context_length: effectiveContextLength }
|
|
196
|
+
: {}),
|
|
197
|
+
...(maxOutputTokens !== undefined ? { max_output_tokens: maxOutputTokens } : {}),
|
|
177
198
|
...(hasLongTier ? { pricing: { overrides: [{ min_prompt_tokens: contextLength }] } } : {}),
|
|
178
199
|
};
|
|
179
200
|
}
|