@bitkyc08/opencodex 2.67.0 → 2.68.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/ocx.mjs +3 -0
- package/gui/dist/assets/App-DCBismRi.css +1 -0
- package/gui/dist/assets/App-mpsibsFj.js +51 -0
- package/gui/dist/assets/Tray-Bn8WhKBH.js +1 -0
- package/gui/dist/assets/{index-DtNmX7hW.css → index-BDUBS8PW.css} +1 -1
- package/gui/dist/assets/index-kUE8tuAK.js +86 -0
- package/gui/dist/assets/quota-summary-DL-bNuNL.js +1 -0
- package/gui/dist/index.html +21 -2
- package/package.json +1 -1
- package/src/adapters/anthropic-image-guard.ts +13 -1
- package/src/adapters/anthropic-image-normalize.ts +3 -2
- package/src/adapters/anthropic-output-schema.ts +36 -0
- package/src/adapters/anthropic.ts +10 -2
- package/src/adapters/base.ts +7 -0
- package/src/adapters/codebuddy/live-models.ts +180 -0
- package/src/adapters/coding-agent/protocol.ts +91 -12
- package/src/adapters/coding-agent/turn.ts +32 -27
- package/src/adapters/devin/cloud-direct/index.ts +1 -0
- package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +18 -4
- package/src/adapters/devin.ts +9 -6
- package/src/adapters/google-http.ts +4 -4
- package/src/adapters/google.ts +93 -3
- package/src/adapters/kiro/adapter.ts +6 -2
- package/src/adapters/kiro/stream.ts +12 -2
- package/src/adapters/kiro/usage.ts +3 -1
- package/src/adapters/kiro-errors.ts +11 -1
- package/src/adapters/kiro-events.ts +27 -1
- package/src/adapters/kiro-refusal.ts +30 -0
- package/src/adapters/kiro-retry.ts +71 -31
- package/src/adapters/openai-chat/deepseek-artifact-schema.ts +45 -0
- package/src/adapters/openai-chat/messages.ts +31 -4
- package/src/adapters/openai-chat/serialized-tool-call-content.ts +52 -7
- package/src/adapters/openai-chat/tool-call-id-remint.ts +65 -0
- package/src/adapters/openai-chat/tool-schema.ts +6 -2
- package/src/adapters/openai-chat.ts +2 -2
- package/src/adapters/openai-responses/muse-tool-choice.ts +31 -0
- package/src/adapters/openai-responses/passthrough.ts +9 -3
- package/src/adapters/opencode-go-additional-tools.ts +19 -10
- package/src/adapters/physical-send.ts +10 -3
- package/src/adapters/registry.ts +2 -1
- package/src/adapters/responses-tool-schema.ts +31 -3
- package/src/adapters/run-turn-queue.ts +63 -23
- package/src/adapters/unique-tool-call-ids.ts +63 -0
- package/src/adapters/xai-web-search.ts +32 -2
- package/src/bridge/sse.ts +14 -2
- package/src/chat/inbound.ts +20 -6
- package/src/claude/agents-inject.ts +4 -0
- package/src/claude/desktop-3p.ts +53 -13
- package/src/claude/desktop-profile.ts +41 -4
- package/src/claude/inbound-model-options.ts +14 -2
- package/src/claude/inbound.ts +1 -1
- package/src/claude/intercept/connect-proxy.ts +17 -1
- package/src/claude/intercept/local-ca.ts +7 -1
- package/src/claude/message-threads.ts +28 -0
- package/src/cli/account-api.ts +13 -0
- package/src/cli/account-auth.ts +49 -2
- package/src/cli/account-extended.ts +4 -2
- package/src/cli/account.ts +4 -2
- package/src/cli/capabilities.ts +17 -0
- package/src/cli/claude.ts +3 -1
- package/src/cli/dispatch.ts +62 -5
- package/src/cli/index.ts +84 -79
- package/src/cli/minimax.ts +4 -2
- package/src/cli/opencode.ts +6 -1
- package/src/cli/restart-handoff.ts +110 -0
- package/src/cli/status.ts +51 -0
- package/src/client/connect.ts +34 -15
- package/src/client/link-ingress.ts +102 -0
- package/src/client/link-join.ts +37 -16
- package/src/client/link-relay.ts +228 -45
- package/src/client/link-state.ts +54 -0
- package/src/client/link-status.ts +39 -0
- package/src/client/link-teardown.ts +2 -2
- package/src/client/link-tunnel.ts +615 -86
- package/src/client/machine-api.ts +10 -4
- package/src/client/machine-listener.ts +57 -10
- package/src/client/runtime.ts +210 -35
- package/src/clients/config-export/constants.ts +1 -12
- package/src/clients/config-export/contracts.ts +2 -0
- package/src/clients/config-export/model-metadata.ts +9 -3
- package/src/clients/config-export/omp.ts +1 -1
- package/src/clients/config-export/zcode-store.ts +2 -2
- package/src/clients/config-export.ts +36 -3
- package/src/codex/app-server-processes.ts +32 -0
- package/src/codex/app-server-restart-service.ts +20 -3
- package/src/codex/catalog/access-programs.ts +50 -0
- package/src/codex/catalog/build-entries.ts +13 -0
- package/src/codex/catalog/derive-entry.ts +1 -0
- package/src/codex/catalog/model-hints.ts +11 -0
- package/src/codex/catalog/parsing.ts +6 -2
- package/src/codex/catalog/provider-models.ts +81 -0
- package/src/codex/catalog/retained-sync.ts +3 -0
- package/src/codex/codex-write-lock.ts +2 -2
- package/src/codex/convergence.ts +2 -0
- package/src/codex/desired-state.ts +14 -4
- package/src/codex/home.ts +21 -3
- package/src/codex/inject/restore.ts +17 -0
- package/src/codex/inject/routing-classify.ts +3 -2
- package/src/codex/inject-coordination.ts +4 -1
- package/src/codex/inject.ts +8 -1
- package/src/codex/journal.ts +6 -2
- package/src/codex/management-convergence.ts +9 -0
- package/src/codex/model-entitlements.ts +28 -5
- package/src/codex/routing/idle-window.ts +58 -0
- package/src/codex/routing-drift.ts +134 -0
- package/src/codex/routing-healer.ts +419 -0
- package/src/codex/routing.ts +10 -0
- package/src/codex/runtime.ts +188 -57
- package/src/codex/sibling-handoff.ts +71 -0
- package/src/codex/sibling-start.ts +139 -0
- package/src/codex/sync.ts +5 -3
- package/src/combos/failover.ts +32 -5
- package/src/combos/request.ts +11 -3
- package/src/combos/reset-window.ts +10 -5
- package/src/combos/resolve.ts +36 -6
- package/src/config/diagnostics.ts +2 -1
- package/src/config/load-degrade.ts +2 -0
- package/src/config/paths.ts +7 -1
- package/src/config/process-state.ts +10 -1
- package/src/config/schema/compaction-recovery.ts +15 -0
- package/src/config/schema/config-schema.ts +2 -0
- package/src/config/schema/leaf-validators.ts +9 -3
- package/src/generated/compatibility-version.json +360 -204
- package/src/images/loop.ts +47 -32
- package/src/integrations/catalog-refresh.ts +4 -0
- package/src/integrations/omp-yaml-source.ts +1 -0
- package/src/lib/claude-request-projection.ts +101 -0
- package/src/lib/local-account-switch-capability.ts +48 -0
- package/src/lib/local-upstream.ts +146 -0
- package/src/lib/package-tree-integrity.ts +25 -3
- package/src/lib/package-tree-retarget.ts +130 -0
- package/src/lib/process-control.ts +4 -1
- package/src/lib/system-restart-contract.ts +106 -0
- package/src/lib/upstream-retry.ts +57 -1
- package/src/link/ports.ts +9 -0
- package/src/link/ssh-argv.ts +16 -0
- package/src/link/ssh-runner.ts +70 -2
- package/src/link/tunnel-state.ts +58 -9
- package/src/oauth/account-quota-rank.ts +29 -18
- package/src/oauth/generic-account-failover.ts +179 -22
- package/src/oauth/index.ts +17 -4
- package/src/oauth/kiro-account-load.ts +107 -0
- package/src/oauth/kiro-device-login.ts +312 -0
- package/src/oauth/kiro-terminal-failover.ts +28 -0
- package/src/oauth/login-flow-state.ts +7 -1
- package/src/oauth/pool-settings-capability.ts +23 -14
- package/src/oauth/store.ts +56 -3
- package/src/oauth/types.ts +4 -0
- package/src/plugins/loader.ts +350 -0
- package/src/plugins/upstream-hooks.ts +119 -0
- package/src/protocols/encoders/adapter-events.ts +9 -2
- package/src/providers/account-quota-disk.ts +42 -4
- package/src/providers/codebuddy-models.ts +6 -2
- package/src/providers/command-code-efforts.ts +33 -10
- package/src/providers/kiro-account-state-disk.ts +44 -0
- package/src/providers/kiro-model-catalog.ts +162 -0
- package/src/providers/kiro-models.ts +5 -4
- package/src/providers/kiro-quota-metrics.ts +35 -0
- package/src/providers/kiro-usage.ts +118 -14
- package/src/providers/quota/account-cache.ts +43 -7
- package/src/providers/quota/antigravity.ts +8 -3
- package/src/providers/quota/kiro-account-probe.ts +12 -0
- package/src/providers/quota/vendor-probes-key.ts +31 -26
- package/src/providers/quota/vendor-probes-oauth.ts +16 -7
- package/src/providers/quota-types.ts +14 -0
- package/src/providers/quota.ts +21 -21
- package/src/providers/registry/entries-extended.ts +5 -3
- package/src/providers/request-pacing.ts +410 -12
- package/src/remote-control/workspace-codex-runtime.ts +4 -3
- package/src/responses/citation-markers.ts +132 -65
- package/src/responses/hosted-tool-policy.ts +14 -3
- package/src/responses/parser-content.ts +3 -2
- package/src/responses/parser.ts +4 -1
- package/src/responses/schema.ts +4 -0
- package/src/responses/visualization-directives.ts +182 -0
- package/src/server/chat-native-sse.ts +24 -12
- package/src/server/chat-native.ts +137 -68
- package/src/server/claude-messages.ts +147 -12
- package/src/server/index/link-listener.ts +7 -0
- package/src/server/index/optional-listeners.ts +4 -1
- package/src/server/index/package-tree-guard.ts +35 -16
- package/src/server/index/serve-options.ts +4 -2
- package/src/server/index/startup-warnings.ts +6 -0
- package/src/server/index.ts +7 -7
- package/src/server/inference/client-encoder-delivery.ts +3 -0
- package/src/server/inference/context.ts +30 -2
- package/src/server/local-account-switch-auth.ts +79 -0
- package/src/server/management/agent-settings-routes.ts +11 -3
- package/src/server/management/config-routes.ts +28 -11
- package/src/server/management/context.ts +2 -0
- package/src/server/management/link-routes.ts +123 -39
- package/src/server/management/logs-usage-routes.ts +9 -42
- package/src/server/management/model-rows.ts +1 -0
- package/src/server/management/oauth-account-routes.ts +67 -12
- package/src/server/management/route-registry.ts +5 -5
- package/src/server/management/sibling-guard.ts +60 -0
- package/src/server/management/storage-log-guard-routes.ts +72 -6
- package/src/server/management/system-restart.ts +92 -135
- package/src/server/management/system-routes.ts +1 -1
- package/src/server/management-api.ts +31 -3
- package/src/server/management-auth.ts +3 -0
- package/src/server/port-reclaim.ts +111 -1
- package/src/server/proxy-liveness.ts +16 -0
- package/src/server/relay.ts +16 -15
- package/src/server/request-log-filter.ts +28 -27
- package/src/server/request-log.ts +7 -3
- package/src/server/request-metrics.ts +28 -0
- package/src/server/responses/adapter-continuation.ts +151 -21
- package/src/server/responses/adapter-delivery.ts +21 -7
- package/src/server/responses/adapter-dispatch.ts +239 -51
- package/src/server/responses/codex-ws-pool.ts +5 -4
- package/src/server/responses/codex-ws-request.ts +4 -1
- package/src/server/responses/compact.ts +8 -4
- package/src/server/responses/compaction-recovery-policy.ts +116 -0
- package/src/server/responses/compaction-recovery.ts +330 -0
- package/src/server/responses/core-combo-failure.ts +1 -0
- package/src/server/responses/core-combo.ts +5 -1
- package/src/server/responses/core-lifetime.ts +22 -0
- package/src/server/responses/core-options.ts +16 -1
- package/src/server/responses/core.ts +14 -4
- package/src/server/responses/empty-completion-guard.ts +2 -0
- package/src/server/responses/fetch-helpers.ts +82 -31
- package/src/server/responses/passthrough-delivery.ts +3 -2
- package/src/server/responses/passthrough-dispatch.ts +12 -2
- package/src/server/responses/passthrough-execution.ts +2 -2
- package/src/server/responses/request-send-budget.ts +21 -5
- package/src/server/responses/request-transport.ts +116 -18
- package/src/server/responses/reset-replay.ts +32 -0
- package/src/server/responses/run-turn-execution.ts +136 -19
- package/src/server/responses/sidecar-execution.ts +41 -8
- package/src/server/responses/terminal-guard.ts +2 -0
- package/src/server/responses/ws-upstream.ts +44 -4
- package/src/server/restart-replacement.ts +382 -0
- package/src/server/startup-health-cache.ts +23 -4
- package/src/server/stop-teardown.ts +35 -1
- package/src/server/system-env.ts +3 -0
- package/src/service/windows-taskxml.ts +11 -7
- package/src/service/windows-wrapper-exit.ts +8 -0
- package/src/stall-timeout.ts +35 -10
- package/src/storage/policy-job.ts +4 -0
- package/src/storage/scanner.ts +177 -29
- package/src/storage/storage-mutation-coordinator.ts +17 -0
- package/src/tray/windows-tray.ps1 +50 -23
- package/src/types/config.ts +4 -0
- package/src/types/provider.ts +5 -1
- package/src/types/request.ts +13 -2
- package/src/update/index.ts +3 -1
- package/src/update/job.ts +6 -2
- package/src/update/mise-launcher-target.ts +100 -0
- package/src/update/worker-launch.ts +46 -0
- package/src/usage/log.ts +2 -0
- package/src/web-search/loop.ts +5 -2
- package/gui/dist/assets/App-BJsT8Icc.css +0 -1
- package/gui/dist/assets/App-D3pNGiN4.js +0 -51
- package/gui/dist/assets/Tray-yb05wCjb.js +0 -1
- package/gui/dist/assets/index-CouvvtMV.js +0 -86
- package/gui/dist/assets/tray-data-CYGdjCJ7.js +0 -1
|
@@ -1,8 +1,34 @@
|
|
|
1
1
|
import type { OcxProviderConfig, RequestPacingRule } from "../types";
|
|
2
2
|
import type { GenerationContext } from "../lib/state-store-sweeper";
|
|
3
|
+
import { carryReplayRefusal, isNonReplayableResponse, markResponseNonReplayable } from "../lib/upstream-retry";
|
|
4
|
+
import { redactSecretString } from "../lib/redact";
|
|
3
5
|
|
|
4
6
|
export const REQUEST_PACING_MAX_QUEUE_DEPTH = 256;
|
|
5
7
|
export const REQUEST_PACING_MAX_QUEUE_AGE_MS = 60_000;
|
|
8
|
+
/**
|
|
9
|
+
* A leased response body that is neither read nor cancelled for this long releases its
|
|
10
|
+
* lease and cancels the body: a caller that dropped a Response without touching its body
|
|
11
|
+
* must not hold a concurrency slot until process restart. The first pull or cancel swaps
|
|
12
|
+
* this deadline for the longer inactivity window below.
|
|
13
|
+
*/
|
|
14
|
+
export const REQUEST_PACING_UNCONSUMED_BODY_MS = 30_000;
|
|
15
|
+
/**
|
|
16
|
+
* Once a tracked body has been pulled at least once, this much silence reclaims its lease
|
|
17
|
+
* and cancels the body: a consumer that reads part of a response and then abandons it
|
|
18
|
+
* without cancel() (a clone-based peek that cancels only its own tee branch) must not
|
|
19
|
+
* hold a concurrency slot forever either. The window is far longer than the unconsumed
|
|
20
|
+
* deadline because a live consumer waiting on a slow upstream is indistinguishable from
|
|
21
|
+
* an abandoned one at this layer; providers keep SSE connections alive with keep-alives
|
|
22
|
+
* well inside this window.
|
|
23
|
+
*/
|
|
24
|
+
export const REQUEST_PACING_BODY_INACTIVITY_MS = 300_000;
|
|
25
|
+
/**
|
|
26
|
+
* Retry-After floor for a waiter refused while blocked by a concurrency cap rather than
|
|
27
|
+
* by a spacing interval. Interval readiness computes to zero under a pure cap, and the
|
|
28
|
+
* 1s minimum would invite clients to re-hit a saturated provider every second, which is
|
|
29
|
+
* the shared-account 429 churn this feature exists to prevent.
|
|
30
|
+
*/
|
|
31
|
+
export const REQUEST_PACING_CONCURRENCY_RETRY_AFTER_SECONDS = 5;
|
|
6
32
|
|
|
7
33
|
let maxQueueDepth = REQUEST_PACING_MAX_QUEUE_DEPTH;
|
|
8
34
|
let maxQueueAgeMs = REQUEST_PACING_MAX_QUEUE_AGE_MS;
|
|
@@ -33,9 +59,11 @@ interface Waiter {
|
|
|
33
59
|
modelId?: string;
|
|
34
60
|
providerIntervalMs: number;
|
|
35
61
|
modelIntervalMs: number;
|
|
62
|
+
providerMaxConcurrent: number;
|
|
63
|
+
modelMaxConcurrent: number;
|
|
36
64
|
queuedAt: number;
|
|
37
65
|
signal?: AbortSignal;
|
|
38
|
-
resolve: () => void;
|
|
66
|
+
resolve: (slot: ProviderRequestSlot) => void;
|
|
39
67
|
reject: (reason: unknown) => void;
|
|
40
68
|
abort?: () => void;
|
|
41
69
|
}
|
|
@@ -44,11 +72,51 @@ interface ProviderPacer {
|
|
|
44
72
|
queue: Waiter[];
|
|
45
73
|
providerNextStartAt: number;
|
|
46
74
|
modelNextStartAt: Map<string, number>;
|
|
75
|
+
providerInFlight: number;
|
|
76
|
+
modelInFlight: Map<string, number>;
|
|
47
77
|
timer?: unknown;
|
|
48
78
|
lastStartedAt?: number;
|
|
49
79
|
lastModelId?: string;
|
|
50
80
|
}
|
|
51
81
|
|
|
82
|
+
/**
|
|
83
|
+
* Lease returned by waitForProviderRequestSlot. Inert unless the provider (or the
|
|
84
|
+
* request model override) sets maxConcurrentRequests; then release() returns the
|
|
85
|
+
* concurrency slot when the upstream request finishes, whether that is body completion,
|
|
86
|
+
* body cancellation, or a send that never produced a response.
|
|
87
|
+
*/
|
|
88
|
+
export interface ProviderRequestSlot {
|
|
89
|
+
/** Provider identity for operator diagnostics: names the lease holder in deadline warnings. */
|
|
90
|
+
readonly providerName?: string;
|
|
91
|
+
/**
|
|
92
|
+
* True only when this slot holds a concurrency lease whose release must follow the
|
|
93
|
+
* upstream body lifecycle. Interval-only slots stay inert so response objects keep
|
|
94
|
+
* their identity through the fetch path.
|
|
95
|
+
*/
|
|
96
|
+
readonly leased: boolean;
|
|
97
|
+
/**
|
|
98
|
+
* True once a tracked response body owns this lease, meaning body completion, body
|
|
99
|
+
* cancellation, or the unconsumed-body deadline will release it. Turn-end cleanup must
|
|
100
|
+
* then leave the release to that lifecycle instead of returning the lease early.
|
|
101
|
+
*/
|
|
102
|
+
readonly bodyTracked: boolean;
|
|
103
|
+
/**
|
|
104
|
+
* True once release() has run. A turn-scoped transport uses this to stop pacing by
|
|
105
|
+
* interval after the lease it relied on is gone (a send that threw, a body that
|
|
106
|
+
* closed) and re-acquire with concurrency, so the cap keeps counting its follow-ups.
|
|
107
|
+
*/
|
|
108
|
+
readonly released: boolean;
|
|
109
|
+
/** Idempotent. Safe to call from body completion, cancellation, and error paths alike. */
|
|
110
|
+
release(): void;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** Internal: leased slots expose this to trackProviderRequestSlotBody when a body takes over the release. */
|
|
114
|
+
interface BodyTrackableProviderRequestSlot extends ProviderRequestSlot {
|
|
115
|
+
markBodyTracked(): void;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
const inertProviderRequestSlot: ProviderRequestSlot = { leased: false, bodyTracked: false, released: false, release() {} };
|
|
119
|
+
|
|
52
120
|
export interface RequestPacingRuntime {
|
|
53
121
|
now: () => number;
|
|
54
122
|
setTimer: (callback: () => void, delayMs: number) => unknown;
|
|
@@ -61,6 +129,7 @@ export interface ProviderRequestPacingStatus {
|
|
|
61
129
|
enabled: boolean;
|
|
62
130
|
queued: number;
|
|
63
131
|
nextSlotInMs: number;
|
|
132
|
+
inFlight?: number;
|
|
64
133
|
lastStartedAt?: number;
|
|
65
134
|
lastModelId?: string;
|
|
66
135
|
}
|
|
@@ -90,6 +159,15 @@ function normalizedInterval(rule: RequestPacingRule | undefined): number {
|
|
|
90
159
|
return Math.max(rpmInterval, fixedInterval);
|
|
91
160
|
}
|
|
92
161
|
|
|
162
|
+
function normalizedMaxConcurrent(rule: RequestPacingRule | undefined): number {
|
|
163
|
+
return typeof rule?.maxConcurrentRequests === "number" && rule.maxConcurrentRequests > 0
|
|
164
|
+
// Floor so runtime-injected configs that bypass the integer schema cannot admit
|
|
165
|
+
// one request past the configured ceiling (a 2.5 cap must not let a third start);
|
|
166
|
+
// clamp sub-1 fractionals up to 1 so a positive cap never degrades into none.
|
|
167
|
+
? Math.max(1, Math.floor(rule.maxConcurrentRequests))
|
|
168
|
+
: 0;
|
|
169
|
+
}
|
|
170
|
+
|
|
93
171
|
export function requestPacingIntervalMs(provider: OcxProviderConfig, modelId?: string): number {
|
|
94
172
|
const policy = provider.requestPacing;
|
|
95
173
|
if (!policy?.enabled) return 0;
|
|
@@ -100,15 +178,34 @@ export function requestPacingIntervalMs(provider: OcxProviderConfig, modelId?: s
|
|
|
100
178
|
function requestPacingIntervals(provider: OcxProviderConfig, modelId?: string): {
|
|
101
179
|
providerIntervalMs: number;
|
|
102
180
|
modelIntervalMs: number;
|
|
181
|
+
providerMaxConcurrent: number;
|
|
182
|
+
modelMaxConcurrent: number;
|
|
103
183
|
} {
|
|
104
184
|
const policy = provider.requestPacing;
|
|
105
|
-
if (!policy?.enabled)
|
|
185
|
+
if (!policy?.enabled) {
|
|
186
|
+
return { providerIntervalMs: 0, modelIntervalMs: 0, providerMaxConcurrent: 0, modelMaxConcurrent: 0 };
|
|
187
|
+
}
|
|
106
188
|
return {
|
|
107
189
|
providerIntervalMs: normalizedInterval(policy),
|
|
108
190
|
modelIntervalMs: modelId ? normalizedInterval(policy.models?.[modelId]) : 0,
|
|
191
|
+
providerMaxConcurrent: normalizedMaxConcurrent(policy),
|
|
192
|
+
modelMaxConcurrent: modelId ? normalizedMaxConcurrent(policy.models?.[modelId]) : 0,
|
|
109
193
|
};
|
|
110
194
|
}
|
|
111
195
|
|
|
196
|
+
/**
|
|
197
|
+
* Whether a concurrency cap applies to this provider/model, expressed as the looser of
|
|
198
|
+
* the two configured bounds. Enforcement gates the provider cap and the model cap
|
|
199
|
+
* independently (a model override of 1 inside a provider cap of 5 admits one), so the
|
|
200
|
+
* return value is a cap-presence signal — every current caller tests it for > 0 — and
|
|
201
|
+
* NOT an effective ceiling. A caller that needs the true ceiling must compute
|
|
202
|
+
* min(provider, model || Infinity) itself.
|
|
203
|
+
*/
|
|
204
|
+
export function requestPacingMaxConcurrentRequests(provider: OcxProviderConfig, modelId?: string): number {
|
|
205
|
+
const limits = requestPacingIntervals(provider, modelId);
|
|
206
|
+
return Math.max(limits.providerMaxConcurrent, limits.modelMaxConcurrent);
|
|
207
|
+
}
|
|
208
|
+
|
|
112
209
|
function waiterReadyAt(state: ProviderPacer, modelId: string | undefined): number {
|
|
113
210
|
return Math.max(
|
|
114
211
|
state.providerNextStartAt,
|
|
@@ -116,8 +213,65 @@ function waiterReadyAt(state: ProviderPacer, modelId: string | undefined): numbe
|
|
|
116
213
|
);
|
|
117
214
|
}
|
|
118
215
|
|
|
119
|
-
function pacingRetryAfterSeconds(
|
|
120
|
-
|
|
216
|
+
function pacingRetryAfterSeconds(
|
|
217
|
+
state: ProviderPacer,
|
|
218
|
+
modelId: string | undefined,
|
|
219
|
+
now: number,
|
|
220
|
+
concurrencyCapped = false,
|
|
221
|
+
): number {
|
|
222
|
+
const intervalSeconds = Math.max(1, Math.ceil(Math.max(0, waiterReadyAt(state, modelId) - now) / 1000));
|
|
223
|
+
// A waiter blocked by an in-flight lease has no interval to report: when the wait ends
|
|
224
|
+
// in refusal, the honest answer is that the cap is saturated, not "retry every second".
|
|
225
|
+
return concurrencyCapped
|
|
226
|
+
? Math.max(intervalSeconds, REQUEST_PACING_CONCURRENCY_RETRY_AFTER_SECONDS)
|
|
227
|
+
: intervalSeconds;
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
function makeProviderRequestSlot(
|
|
231
|
+
providerName: string,
|
|
232
|
+
state: ProviderPacer,
|
|
233
|
+
waiter: Waiter,
|
|
234
|
+
): BodyTrackableProviderRequestSlot {
|
|
235
|
+
let released = false;
|
|
236
|
+
let bodyTracked = false;
|
|
237
|
+
const leased = waiter.providerMaxConcurrent > 0 || waiter.modelMaxConcurrent > 0;
|
|
238
|
+
const slot: BodyTrackableProviderRequestSlot = {
|
|
239
|
+
providerName,
|
|
240
|
+
leased,
|
|
241
|
+
get bodyTracked() {
|
|
242
|
+
return bodyTracked;
|
|
243
|
+
},
|
|
244
|
+
get released() {
|
|
245
|
+
return released;
|
|
246
|
+
},
|
|
247
|
+
markBodyTracked() {
|
|
248
|
+
bodyTracked = true;
|
|
249
|
+
},
|
|
250
|
+
release() {
|
|
251
|
+
if (released) return;
|
|
252
|
+
released = true;
|
|
253
|
+
waiter.signal?.removeEventListener("abort", releaseOnAbort);
|
|
254
|
+
if (waiter.providerMaxConcurrent > 0 && state.providerInFlight > 0) state.providerInFlight -= 1;
|
|
255
|
+
if (waiter.modelId && waiter.modelMaxConcurrent > 0) {
|
|
256
|
+
const current = state.modelInFlight.get(waiter.modelId) ?? 0;
|
|
257
|
+
if (current <= 1) state.modelInFlight.delete(waiter.modelId);
|
|
258
|
+
else state.modelInFlight.set(waiter.modelId, current - 1);
|
|
259
|
+
}
|
|
260
|
+
runtime.enqueueMicrotask(() => {
|
|
261
|
+
// A pending wake-up timer makes runQueue defer to it, but the lease that just
|
|
262
|
+
// returned may admit a waiter the timer was never scheduled for, so take over.
|
|
263
|
+
if (state.timer) {
|
|
264
|
+
runtime.clearTimer(state.timer);
|
|
265
|
+
state.timer = undefined;
|
|
266
|
+
}
|
|
267
|
+
runQueue(providerName, state);
|
|
268
|
+
});
|
|
269
|
+
},
|
|
270
|
+
};
|
|
271
|
+
const releaseOnAbort = (): void => slot.release();
|
|
272
|
+
waiter.signal?.addEventListener("abort", releaseOnAbort, { once: true });
|
|
273
|
+
if (waiter.signal?.aborted) slot.release();
|
|
274
|
+
return slot;
|
|
121
275
|
}
|
|
122
276
|
|
|
123
277
|
function rejectExpiredWaiters(providerName: string, state: ProviderPacer, now: number): void {
|
|
@@ -129,7 +283,12 @@ function rejectExpiredWaiters(providerName: string, state: ProviderPacer, now: n
|
|
|
129
283
|
waiter.reject(new RequestPacingQueueOverloadError(
|
|
130
284
|
providerName,
|
|
131
285
|
"queue_expired",
|
|
132
|
-
pacingRetryAfterSeconds(
|
|
286
|
+
pacingRetryAfterSeconds(
|
|
287
|
+
state,
|
|
288
|
+
waiter.modelId,
|
|
289
|
+
now,
|
|
290
|
+
waiter.providerMaxConcurrent > 0 || waiter.modelMaxConcurrent > 0,
|
|
291
|
+
),
|
|
133
292
|
));
|
|
134
293
|
}
|
|
135
294
|
}
|
|
@@ -162,6 +321,9 @@ function runQueue(providerName: string, state: ProviderPacer): void {
|
|
|
162
321
|
|
|
163
322
|
const providerReadyAt = Math.max(now, state.providerNextStartAt);
|
|
164
323
|
const waiterIndex = state.queue.findIndex(waiter => {
|
|
324
|
+
if (waiter.providerMaxConcurrent > 0 && state.providerInFlight >= waiter.providerMaxConcurrent) return false;
|
|
325
|
+
if (waiter.modelId && waiter.modelMaxConcurrent > 0
|
|
326
|
+
&& (state.modelInFlight.get(waiter.modelId) ?? 0) >= waiter.modelMaxConcurrent) return false;
|
|
165
327
|
const modelReadyAt = waiter.modelId ? (state.modelNextStartAt.get(waiter.modelId) ?? 0) : 0;
|
|
166
328
|
return Math.max(providerReadyAt, modelReadyAt) <= now;
|
|
167
329
|
});
|
|
@@ -171,8 +333,15 @@ function runQueue(providerName: string, state: ProviderPacer): void {
|
|
|
171
333
|
const modelReadyAt = waiter.modelId ? (state.modelNextStartAt.get(waiter.modelId) ?? 0) : 0;
|
|
172
334
|
const readyAt = Math.max(providerReadyAt, modelReadyAt);
|
|
173
335
|
const expiresAt = waiter.queuedAt + maxQueueAgeMs;
|
|
174
|
-
|
|
336
|
+
// A waiter whose start time already passed is blocked on an in-flight lease; its release
|
|
337
|
+
// re-runs this queue through a microtask, so the timer only needs its expiry backstop.
|
|
338
|
+
// Scheduling for its readyAt (in the past) would spin the timer on every empty pass.
|
|
339
|
+
earliestAt = Math.min(earliestAt, readyAt <= now ? Number.POSITIVE_INFINITY : readyAt, expiresAt);
|
|
175
340
|
}
|
|
341
|
+
// Unreachable for a non-empty queue: rejectExpiredWaiters above removed every waiter
|
|
342
|
+
// whose expiresAt passed, so each remaining one contributes a finite value. Kept as a
|
|
343
|
+
// belt-and-braces backstop rather than a case future readers should hunt for.
|
|
344
|
+
if (!Number.isFinite(earliestAt)) return;
|
|
176
345
|
const delayMs = Math.max(0, earliestAt - now);
|
|
177
346
|
state.timer = runtime.setTimer(() => {
|
|
178
347
|
state.timer = undefined;
|
|
@@ -190,7 +359,11 @@ function runQueue(providerName: string, state: ProviderPacer): void {
|
|
|
190
359
|
if (waiter.modelId && waiter.modelIntervalMs > 0) {
|
|
191
360
|
state.modelNextStartAt.set(waiter.modelId, startedAt + waiter.modelIntervalMs);
|
|
192
361
|
}
|
|
193
|
-
waiter.
|
|
362
|
+
if (waiter.providerMaxConcurrent > 0) state.providerInFlight += 1;
|
|
363
|
+
if (waiter.modelId && waiter.modelMaxConcurrent > 0) {
|
|
364
|
+
state.modelInFlight.set(waiter.modelId, (state.modelInFlight.get(waiter.modelId) ?? 0) + 1);
|
|
365
|
+
}
|
|
366
|
+
waiter.resolve(makeProviderRequestSlot(providerName, state, waiter));
|
|
194
367
|
runtime.enqueueMicrotask(() => runQueue(providerName, state));
|
|
195
368
|
}
|
|
196
369
|
|
|
@@ -199,13 +372,28 @@ export async function waitForProviderRequestSlot(
|
|
|
199
372
|
provider: OcxProviderConfig,
|
|
200
373
|
modelId?: string,
|
|
201
374
|
signal?: AbortSignal,
|
|
202
|
-
|
|
375
|
+
options?: { concurrency?: boolean },
|
|
376
|
+
): Promise<ProviderRequestSlot> {
|
|
203
377
|
const intervals = requestPacingIntervals(provider, modelId);
|
|
204
|
-
|
|
378
|
+
// A turn-scoped transport sends several physical requests per logical turn (Cursor
|
|
379
|
+
// HTTP/1.1 RunSSE plus BidiAppends). One lease covers the turn; follow-up sends pace
|
|
380
|
+
// by interval only, or a follow-up would queue behind the lease its own turn holds.
|
|
381
|
+
const concurrency = options?.concurrency !== false;
|
|
382
|
+
const waiterIntervals = concurrency ? intervals : {
|
|
383
|
+
providerIntervalMs: intervals.providerIntervalMs,
|
|
384
|
+
modelIntervalMs: intervals.modelIntervalMs,
|
|
385
|
+
providerMaxConcurrent: 0,
|
|
386
|
+
modelMaxConcurrent: 0,
|
|
387
|
+
};
|
|
388
|
+
const paced = Math.max(intervals.providerIntervalMs, intervals.modelIntervalMs) > 0
|
|
389
|
+
|| waiterIntervals.providerMaxConcurrent > 0
|
|
390
|
+
|| waiterIntervals.modelMaxConcurrent > 0;
|
|
391
|
+
if (!paced) return inertProviderRequestSlot;
|
|
205
392
|
if (signal?.aborted) throw abortReason(signal);
|
|
206
393
|
|
|
207
394
|
const state = pacers.get(providerName) ?? {
|
|
208
395
|
queue: [], providerNextStartAt: 0, modelNextStartAt: new Map<string, number>(),
|
|
396
|
+
providerInFlight: 0, modelInFlight: new Map<string, number>(),
|
|
209
397
|
};
|
|
210
398
|
pacers.set(providerName, state);
|
|
211
399
|
|
|
@@ -221,12 +409,24 @@ export async function waitForProviderRequestSlot(
|
|
|
221
409
|
throw new RequestPacingQueueOverloadError(
|
|
222
410
|
providerName,
|
|
223
411
|
"queue_full",
|
|
224
|
-
pacingRetryAfterSeconds(
|
|
412
|
+
pacingRetryAfterSeconds(
|
|
413
|
+
state,
|
|
414
|
+
modelId,
|
|
415
|
+
runtime.now(),
|
|
416
|
+
waiterIntervals.providerMaxConcurrent > 0 || waiterIntervals.modelMaxConcurrent > 0,
|
|
417
|
+
),
|
|
225
418
|
);
|
|
226
419
|
}
|
|
227
420
|
|
|
228
|
-
await new Promise<
|
|
229
|
-
const waiter: Waiter = {
|
|
421
|
+
return await new Promise<ProviderRequestSlot>((resolve, reject) => {
|
|
422
|
+
const waiter: Waiter = {
|
|
423
|
+
modelId,
|
|
424
|
+
...waiterIntervals,
|
|
425
|
+
queuedAt: runtime.now(),
|
|
426
|
+
signal,
|
|
427
|
+
resolve,
|
|
428
|
+
reject,
|
|
429
|
+
};
|
|
230
430
|
waiter.abort = () => {
|
|
231
431
|
const index = state.queue.indexOf(waiter);
|
|
232
432
|
if (index >= 0) state.queue.splice(index, 1);
|
|
@@ -252,6 +452,192 @@ export async function waitForProviderRequestSlot(
|
|
|
252
452
|
});
|
|
253
453
|
}
|
|
254
454
|
|
|
455
|
+
/**
|
|
456
|
+
* Tie a pacing slot lease to an upstream response body: the lease is released when the
|
|
457
|
+
* body completes, errors, or is cancelled by the consumer. A null body (204/304/HEAD)
|
|
458
|
+
* means the exchange is already finished, so the lease returns immediately. A body that
|
|
459
|
+
* is neither read nor cancelled for REQUEST_PACING_UNCONSUMED_BODY_MS releases its lease
|
|
460
|
+
* and cancels the body, so a dropped Response cannot hold a slot forever. The returned
|
|
461
|
+
* Response preserves status, statusText, headers, and the identity-based replay markers.
|
|
462
|
+
*/
|
|
463
|
+
export function trackProviderRequestSlotBody(
|
|
464
|
+
slot: ProviderRequestSlot | undefined,
|
|
465
|
+
response: Response,
|
|
466
|
+
): Response {
|
|
467
|
+
// An unleased slot has nothing to return on body close, and rewrapping the Response
|
|
468
|
+
// would break identity-based markers (the eager WS relay registry is a WeakSet).
|
|
469
|
+
if (!slot?.leased) return response;
|
|
470
|
+
// Slots built outside waitForProviderRequestSlot (test doubles) satisfy only the public
|
|
471
|
+
// ProviderRequestSlot shape; their bodies still own the release through the wrapper below.
|
|
472
|
+
(slot as Partial<BodyTrackableProviderRequestSlot>).markBodyTracked?.();
|
|
473
|
+
if (!response.body) {
|
|
474
|
+
slot.release();
|
|
475
|
+
return response;
|
|
476
|
+
}
|
|
477
|
+
const source = response.body;
|
|
478
|
+
let released = false;
|
|
479
|
+
// Set when a deadline callback cancelled the source: the truncation must surface as a
|
|
480
|
+
// stream error on the next pull, never as a clean EOF a relay would treat as success.
|
|
481
|
+
let expired = false;
|
|
482
|
+
let expiryTimer: unknown;
|
|
483
|
+
let expiryKind: "unconsumed" | "inactive" = "unconsumed";
|
|
484
|
+
const clearExpiryTimer = (): void => {
|
|
485
|
+
if (expiryTimer === undefined) return;
|
|
486
|
+
runtime.clearTimer(expiryTimer);
|
|
487
|
+
expiryTimer = undefined;
|
|
488
|
+
};
|
|
489
|
+
const release = (): void => {
|
|
490
|
+
if (released) return;
|
|
491
|
+
released = true;
|
|
492
|
+
clearExpiryTimer();
|
|
493
|
+
slot.release();
|
|
494
|
+
};
|
|
495
|
+
const armExpiryTimer = (delayMs: number, kind: "unconsumed" | "inactive"): void => {
|
|
496
|
+
clearExpiryTimer();
|
|
497
|
+
expiryKind = kind;
|
|
498
|
+
expiryTimer = runtime.setTimer(() => {
|
|
499
|
+
expiryTimer = undefined;
|
|
500
|
+
if (released) return;
|
|
501
|
+
expired = true;
|
|
502
|
+
release();
|
|
503
|
+
// Cancel through the reader when one exists: after the first pull the source is
|
|
504
|
+
// locked to it, and cancelling a locked stream rejects without releasing the
|
|
505
|
+
// socket — the exact upstream leak this deadline exists to prevent.
|
|
506
|
+
void (reader ?? source).cancel().catch(() => {});
|
|
507
|
+
const who = slot.providerName === undefined ? "" : ` for provider ${JSON.stringify(redactSecretString(slot.providerName))}`;
|
|
508
|
+
console.warn(
|
|
509
|
+
expiryKind === "inactive"
|
|
510
|
+
? `[opencodex] requestPacing${who} released a concurrency lease after `
|
|
511
|
+
+ REQUEST_PACING_BODY_INACTIVITY_MS
|
|
512
|
+
+ "ms of response body inactivity; the body was cancelled."
|
|
513
|
+
: `[opencodex] requestPacing${who} released a concurrency lease after `
|
|
514
|
+
+ REQUEST_PACING_UNCONSUMED_BODY_MS
|
|
515
|
+
+ "ms because the provider response body was neither read nor cancelled; the body was cancelled.",
|
|
516
|
+
);
|
|
517
|
+
}, delayMs);
|
|
518
|
+
};
|
|
519
|
+
armExpiryTimer(REQUEST_PACING_UNCONSUMED_BODY_MS, "unconsumed");
|
|
520
|
+
let reader: ReadableStreamDefaultReader<Uint8Array> | undefined;
|
|
521
|
+
let cancelled = false;
|
|
522
|
+
const tracked = new ReadableStream<Uint8Array>({
|
|
523
|
+
pull: async controller => {
|
|
524
|
+
// A pull proves a consumer is attached, not that it will keep reading: re-arm a
|
|
525
|
+
// longer inactivity deadline instead of disarming. A body read once and then
|
|
526
|
+
// abandoned (a clone-based peek that cancels only its own tee branch) must still
|
|
527
|
+
// return its lease, while a live stream keeps pushing the deadline back per pull.
|
|
528
|
+
if (!released) armExpiryTimer(REQUEST_PACING_BODY_INACTIVITY_MS, "inactive");
|
|
529
|
+
try {
|
|
530
|
+
// Inside the try: a source another reader already locked makes getReader()
|
|
531
|
+
// throw, and with the deadline disarmed above that failure must release the
|
|
532
|
+
// lease here instead of stranding it for the process lifetime.
|
|
533
|
+
reader ??= source.getReader();
|
|
534
|
+
const { done, value } = await reader.read();
|
|
535
|
+
// A consumer cancel while this read was pending resolves it (done or a late chunk);
|
|
536
|
+
// touching the cancelled controller would throw from the pull algorithm.
|
|
537
|
+
if (cancelled) return;
|
|
538
|
+
if (expired) {
|
|
539
|
+
// The deadline cancelled the source mid-stream; report the truncation as an
|
|
540
|
+
// error instead of letting the cancelled read's done flag close the stream.
|
|
541
|
+
controller.error(new Error(
|
|
542
|
+
"[opencodex] requestPacing cancelled the response body after its lease deadline",
|
|
543
|
+
));
|
|
544
|
+
return;
|
|
545
|
+
}
|
|
546
|
+
if (done) {
|
|
547
|
+
controller.close();
|
|
548
|
+
release();
|
|
549
|
+
return;
|
|
550
|
+
}
|
|
551
|
+
controller.enqueue(value);
|
|
552
|
+
} catch (error) {
|
|
553
|
+
if (cancelled) return;
|
|
554
|
+
release();
|
|
555
|
+
controller.error(error);
|
|
556
|
+
}
|
|
557
|
+
},
|
|
558
|
+
cancel: reason => {
|
|
559
|
+
cancelled = true;
|
|
560
|
+
clearExpiryTimer();
|
|
561
|
+
release();
|
|
562
|
+
return (reader ?? source).cancel(reason);
|
|
563
|
+
},
|
|
564
|
+
}, {
|
|
565
|
+
// A default-count stream pulls once at construction with no reader attached, which
|
|
566
|
+
// would mark the body consumed and disarm the deadline before any real consumer
|
|
567
|
+
// arrives. Zero capacity keeps pull consumer-driven: the first read arms nothing
|
|
568
|
+
// until a reader actually asks for bytes.
|
|
569
|
+
highWaterMark: 0,
|
|
570
|
+
});
|
|
571
|
+
let wrapped: Response;
|
|
572
|
+
try {
|
|
573
|
+
wrapped = new Response(tracked, {
|
|
574
|
+
status: response.status,
|
|
575
|
+
statusText: response.statusText,
|
|
576
|
+
headers: response.headers,
|
|
577
|
+
});
|
|
578
|
+
} catch {
|
|
579
|
+
// A non-conforming status (a proxy passing a raw 6xx through) or a body on a
|
|
580
|
+
// null-body status throws here after markBodyTracked, and boundary cleanup skips
|
|
581
|
+
// body-tracked slots: return the lease now instead of waiting out the deadline.
|
|
582
|
+
// The abandoned rewrap never locked the source (no pull ran), so hand back the
|
|
583
|
+
// ORIGINAL response with its body intact: rethrowing would make the google-http,
|
|
584
|
+
// command-code and mimo retry ladders replay a request whose response did arrive.
|
|
585
|
+
release();
|
|
586
|
+
return response;
|
|
587
|
+
}
|
|
588
|
+
// Retry helpers mark the response they return from, and recovery decisions key on these
|
|
589
|
+
// identity markers; the rewrap must not make a non-replayable response look replayable.
|
|
590
|
+
if (isNonReplayableResponse(response)) markResponseNonReplayable(wrapped);
|
|
591
|
+
return carryReplayRefusal(response, wrapped);
|
|
592
|
+
}
|
|
593
|
+
|
|
594
|
+
/**
|
|
595
|
+
* Return a lease at a turn or attempt boundary: a lease a tracked response body now owns
|
|
596
|
+
* is left to that body's lifecycle, and any other unconsumed lease is released. Idempotent,
|
|
597
|
+
* and a no-op for interval-only slots, so every boundary can call it unconditionally.
|
|
598
|
+
*/
|
|
599
|
+
export function releaseProviderRequestSlot(slot: ProviderRequestSlot | undefined): void {
|
|
600
|
+
if (!slot || slot.bodyTracked) return;
|
|
601
|
+
slot.release();
|
|
602
|
+
}
|
|
603
|
+
|
|
604
|
+
/** Transfer a physical send's lease to its response body, or release it on failure. */
|
|
605
|
+
export async function sendTrackingRequestSlot(
|
|
606
|
+
slot: ProviderRequestSlot | undefined,
|
|
607
|
+
send: () => Promise<Response>,
|
|
608
|
+
): Promise<Response> {
|
|
609
|
+
try {
|
|
610
|
+
return trackProviderRequestSlotBody(slot, await send());
|
|
611
|
+
} catch (error) {
|
|
612
|
+
slot?.release();
|
|
613
|
+
throw error;
|
|
614
|
+
}
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
/**
|
|
618
|
+
* Acquire one pacing lease and return it at the boundary. The send callback runs with
|
|
619
|
+
* the lease; this helper releases it unless a tracked response body has taken ownership.
|
|
620
|
+
* A send that throws (an abort, a send-budget refusal, a build failure) or returns
|
|
621
|
+
* without ever dispatching through the executor would otherwise strand the lease for
|
|
622
|
+
* the process lifetime, and a dispatched send's tracked body keeps its own release.
|
|
623
|
+
* Centralized so every dispatch boundary shares one copy of the acquire/release pairing
|
|
624
|
+
* and a future edit cannot fork the lease lifecycle.
|
|
625
|
+
*/
|
|
626
|
+
export async function withProviderRequestSlot<T>(
|
|
627
|
+
providerName: string,
|
|
628
|
+
provider: OcxProviderConfig,
|
|
629
|
+
modelId: string | undefined,
|
|
630
|
+
signal: AbortSignal | undefined,
|
|
631
|
+
send: (pacingSlot: ProviderRequestSlot) => Promise<T>,
|
|
632
|
+
): Promise<T> {
|
|
633
|
+
const pacingSlot = await waitForProviderRequestSlot(providerName, provider, modelId, signal);
|
|
634
|
+
try {
|
|
635
|
+
return await send(pacingSlot);
|
|
636
|
+
} finally {
|
|
637
|
+
releaseProviderRequestSlot(pacingSlot);
|
|
638
|
+
}
|
|
639
|
+
}
|
|
640
|
+
|
|
255
641
|
export function providerRequestPacingStatus(
|
|
256
642
|
providerName: string,
|
|
257
643
|
provider: OcxProviderConfig,
|
|
@@ -266,11 +652,23 @@ export function providerRequestPacingStatus(
|
|
|
266
652
|
}
|
|
267
653
|
if (Number.isFinite(earliestQueuedSlotAt)) nextSlotAt = earliestQueuedSlotAt;
|
|
268
654
|
}
|
|
655
|
+
const providerConcurrencyCap = requestPacingMaxConcurrentRequests(provider);
|
|
656
|
+
// Gate on enabled so the status surface agrees with enforcement: a disabled requestPacing
|
|
657
|
+
// block must not report in-flight leases alongside enabled: false.
|
|
658
|
+
const anyConcurrencyCap = provider.requestPacing?.enabled === true && (providerConcurrencyCap > 0
|
|
659
|
+
|| Object.values(provider.requestPacing?.models ?? {}).some(rule => normalizedMaxConcurrent(rule) > 0));
|
|
269
660
|
return {
|
|
270
661
|
provider: providerName,
|
|
271
662
|
enabled: provider.requestPacing?.enabled === true,
|
|
272
663
|
queued: state?.queue.length ?? 0,
|
|
273
664
|
nextSlotInMs: Math.max(0, Math.ceil(nextSlotAt - now)),
|
|
665
|
+
// A provider cap counts every paced send in providerInFlight; model-only caps keep
|
|
666
|
+
// providerInFlight at zero, so report the per-model sum for those providers instead.
|
|
667
|
+
...(anyConcurrencyCap ? {
|
|
668
|
+
inFlight: providerConcurrencyCap > 0
|
|
669
|
+
? state?.providerInFlight ?? 0
|
|
670
|
+
: [...(state?.modelInFlight.values() ?? [])].reduce((sum, count) => sum + count, 0),
|
|
671
|
+
} : {}),
|
|
274
672
|
...(state?.lastStartedAt !== undefined ? { lastStartedAt: state.lastStartedAt } : {}),
|
|
275
673
|
...(state?.lastModelId ? { lastModelId: state.lastModelId } : {}),
|
|
276
674
|
};
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { chmodSync, linkSync, mkdirSync, mkdtempSync, realpathSync, symlinkSync } from "node:fs";
|
|
2
2
|
import { tmpdir } from "node:os";
|
|
3
3
|
import { dirname, isAbsolute, join } from "node:path";
|
|
4
|
-
import {
|
|
4
|
+
import { resolveCodexRuntimeAsync } from "../codex/runtime";
|
|
5
5
|
import { remoteWorkspaceThreadStartParams } from "./workspace-coordinator";
|
|
6
6
|
import { startRemoteWorkspaceToolBridge } from "./workspace-tool-bridge";
|
|
7
7
|
import { truncateRemoteWorkspaceUtf8 } from "./workspace-utf8";
|
|
@@ -299,7 +299,8 @@ export class CodexRemoteWorkspaceRuntimeFactory implements RemoteWorkspaceRuntim
|
|
|
299
299
|
if (this.options.command && this.options.command.length > 0) {
|
|
300
300
|
return { available: true, version: this.options.version ?? "test" };
|
|
301
301
|
}
|
|
302
|
-
|
|
302
|
+
// Async probes: a Hub availability check must not freeze every other proxy request.
|
|
303
|
+
const resolved = await resolveCodexRuntimeAsync();
|
|
303
304
|
const compatibility = codexRemotePermissionProfileCompatibility();
|
|
304
305
|
if (!compatibility.compatible) return { available: false, reason: compatibility.reason };
|
|
305
306
|
return resolved.runtime.version
|
|
@@ -310,7 +311,7 @@ export class CodexRemoteWorkspaceRuntimeFactory implements RemoteWorkspaceRuntim
|
|
|
310
311
|
async start(options: Parameters<RemoteWorkspaceRuntimeFactory["start"]>[0]): Promise<RemoteWorkspaceRuntimeHandle> {
|
|
311
312
|
const command = this.options.command
|
|
312
313
|
? [...this.options.command]
|
|
313
|
-
: [
|
|
314
|
+
: [(await resolveCodexRuntimeAsync()).runtime.command];
|
|
314
315
|
if (command.length < 1) throw new Error("Codex CLI is unavailable on this Hub");
|
|
315
316
|
const executablePath = isAbsolute(command[0]!) ? command[0]! : findExecutableOnPath(command[0]!);
|
|
316
317
|
if (!executablePath) throw new Error("Codex CLI executable could not be resolved on this Hub");
|