@bitkyc08/opencodex 2.67.0 → 2.68.0-preview.20260927

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (257) hide show
  1. package/bin/ocx.mjs +3 -0
  2. package/gui/dist/assets/App-B5-cAEd2.js +51 -0
  3. package/gui/dist/assets/App-DCBismRi.css +1 -0
  4. package/gui/dist/assets/Tray-DFnMiyD0.js +1 -0
  5. package/gui/dist/assets/{index-DtNmX7hW.css → index-BDUBS8PW.css} +1 -1
  6. package/gui/dist/assets/index-BcXblvem.js +86 -0
  7. package/gui/dist/assets/quota-summary-i84eOU3o.js +1 -0
  8. package/gui/dist/index.html +21 -2
  9. package/package.json +1 -1
  10. package/src/adapters/anthropic-image-guard.ts +13 -1
  11. package/src/adapters/anthropic-image-normalize.ts +3 -2
  12. package/src/adapters/anthropic-output-schema.ts +36 -0
  13. package/src/adapters/anthropic.ts +10 -2
  14. package/src/adapters/base.ts +7 -0
  15. package/src/adapters/codebuddy/live-models.ts +180 -0
  16. package/src/adapters/coding-agent/protocol.ts +91 -12
  17. package/src/adapters/coding-agent/turn.ts +32 -27
  18. package/src/adapters/devin/cloud-direct/index.ts +1 -0
  19. package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +18 -4
  20. package/src/adapters/devin.ts +9 -6
  21. package/src/adapters/google-http.ts +4 -4
  22. package/src/adapters/google.ts +93 -3
  23. package/src/adapters/kiro/adapter.ts +6 -2
  24. package/src/adapters/kiro/stream.ts +12 -2
  25. package/src/adapters/kiro/usage.ts +3 -1
  26. package/src/adapters/kiro-errors.ts +11 -1
  27. package/src/adapters/kiro-events.ts +27 -1
  28. package/src/adapters/kiro-refusal.ts +30 -0
  29. package/src/adapters/kiro-retry.ts +71 -31
  30. package/src/adapters/openai-chat/deepseek-artifact-schema.ts +45 -0
  31. package/src/adapters/openai-chat/messages.ts +31 -4
  32. package/src/adapters/openai-chat/serialized-tool-call-content.ts +52 -7
  33. package/src/adapters/openai-chat/tool-call-id-remint.ts +65 -0
  34. package/src/adapters/openai-chat/tool-schema.ts +6 -2
  35. package/src/adapters/openai-chat.ts +2 -2
  36. package/src/adapters/openai-responses/muse-tool-choice.ts +31 -0
  37. package/src/adapters/openai-responses/passthrough.ts +9 -3
  38. package/src/adapters/opencode-go-additional-tools.ts +19 -10
  39. package/src/adapters/physical-send.ts +10 -3
  40. package/src/adapters/registry.ts +2 -1
  41. package/src/adapters/responses-tool-schema.ts +31 -3
  42. package/src/adapters/run-turn-queue.ts +63 -23
  43. package/src/adapters/unique-tool-call-ids.ts +63 -0
  44. package/src/adapters/xai-web-search.ts +32 -2
  45. package/src/bridge/sse.ts +14 -2
  46. package/src/chat/inbound.ts +20 -6
  47. package/src/claude/agents-inject.ts +4 -0
  48. package/src/claude/desktop-3p.ts +53 -13
  49. package/src/claude/desktop-profile.ts +41 -4
  50. package/src/claude/inbound-model-options.ts +14 -2
  51. package/src/claude/inbound.ts +1 -1
  52. package/src/claude/intercept/connect-proxy.ts +17 -1
  53. package/src/claude/intercept/local-ca.ts +7 -1
  54. package/src/claude/message-threads.ts +28 -0
  55. package/src/cli/account-api.ts +13 -0
  56. package/src/cli/account-auth.ts +49 -2
  57. package/src/cli/account-extended.ts +4 -2
  58. package/src/cli/account.ts +4 -2
  59. package/src/cli/capabilities.ts +17 -0
  60. package/src/cli/claude.ts +3 -1
  61. package/src/cli/dispatch.ts +62 -5
  62. package/src/cli/index.ts +84 -79
  63. package/src/cli/minimax.ts +4 -2
  64. package/src/cli/opencode.ts +6 -1
  65. package/src/cli/restart-handoff.ts +110 -0
  66. package/src/cli/status.ts +51 -0
  67. package/src/client/connect.ts +34 -15
  68. package/src/client/link-ingress.ts +102 -0
  69. package/src/client/link-join.ts +37 -16
  70. package/src/client/link-relay.ts +228 -45
  71. package/src/client/link-state.ts +54 -0
  72. package/src/client/link-status.ts +39 -0
  73. package/src/client/link-teardown.ts +2 -2
  74. package/src/client/link-tunnel.ts +615 -86
  75. package/src/client/machine-api.ts +10 -4
  76. package/src/client/machine-listener.ts +57 -10
  77. package/src/client/runtime.ts +210 -35
  78. package/src/clients/config-export/constants.ts +1 -12
  79. package/src/clients/config-export/contracts.ts +2 -0
  80. package/src/clients/config-export/model-metadata.ts +9 -3
  81. package/src/clients/config-export/omp.ts +1 -1
  82. package/src/clients/config-export/zcode-store.ts +2 -2
  83. package/src/clients/config-export.ts +36 -3
  84. package/src/codex/app-server-processes.ts +32 -0
  85. package/src/codex/app-server-restart-service.ts +20 -3
  86. package/src/codex/catalog/access-programs.ts +50 -0
  87. package/src/codex/catalog/build-entries.ts +13 -0
  88. package/src/codex/catalog/derive-entry.ts +1 -0
  89. package/src/codex/catalog/model-hints.ts +11 -0
  90. package/src/codex/catalog/parsing.ts +6 -2
  91. package/src/codex/catalog/provider-models.ts +81 -0
  92. package/src/codex/catalog/retained-sync.ts +3 -0
  93. package/src/codex/codex-write-lock.ts +2 -2
  94. package/src/codex/convergence.ts +2 -0
  95. package/src/codex/desired-state.ts +14 -4
  96. package/src/codex/home.ts +21 -3
  97. package/src/codex/inject/restore.ts +17 -0
  98. package/src/codex/inject/routing-classify.ts +3 -2
  99. package/src/codex/inject-coordination.ts +4 -1
  100. package/src/codex/inject.ts +8 -1
  101. package/src/codex/journal.ts +6 -2
  102. package/src/codex/management-convergence.ts +9 -0
  103. package/src/codex/model-entitlements.ts +28 -5
  104. package/src/codex/routing/idle-window.ts +58 -0
  105. package/src/codex/routing-drift.ts +134 -0
  106. package/src/codex/routing-healer.ts +419 -0
  107. package/src/codex/routing.ts +10 -0
  108. package/src/codex/runtime.ts +188 -57
  109. package/src/codex/sibling-handoff.ts +71 -0
  110. package/src/codex/sibling-start.ts +139 -0
  111. package/src/codex/sync.ts +5 -3
  112. package/src/combos/failover.ts +32 -5
  113. package/src/combos/request.ts +11 -3
  114. package/src/combos/reset-window.ts +10 -5
  115. package/src/combos/resolve.ts +36 -6
  116. package/src/config/diagnostics.ts +2 -1
  117. package/src/config/load-degrade.ts +2 -0
  118. package/src/config/paths.ts +7 -1
  119. package/src/config/process-state.ts +10 -1
  120. package/src/config/schema/compaction-recovery.ts +15 -0
  121. package/src/config/schema/config-schema.ts +2 -0
  122. package/src/config/schema/leaf-validators.ts +9 -3
  123. package/src/generated/compatibility-version.json +360 -204
  124. package/src/images/loop.ts +47 -32
  125. package/src/integrations/catalog-refresh.ts +4 -0
  126. package/src/integrations/omp-yaml-source.ts +1 -0
  127. package/src/lib/claude-request-projection.ts +101 -0
  128. package/src/lib/local-account-switch-capability.ts +48 -0
  129. package/src/lib/local-upstream.ts +146 -0
  130. package/src/lib/package-tree-integrity.ts +25 -3
  131. package/src/lib/package-tree-retarget.ts +130 -0
  132. package/src/lib/process-control.ts +4 -1
  133. package/src/lib/system-restart-contract.ts +106 -0
  134. package/src/lib/upstream-retry.ts +57 -1
  135. package/src/link/ports.ts +9 -0
  136. package/src/link/ssh-argv.ts +16 -0
  137. package/src/link/ssh-runner.ts +70 -2
  138. package/src/link/tunnel-state.ts +58 -9
  139. package/src/oauth/account-quota-rank.ts +29 -18
  140. package/src/oauth/generic-account-failover.ts +179 -22
  141. package/src/oauth/index.ts +17 -4
  142. package/src/oauth/kiro-account-load.ts +107 -0
  143. package/src/oauth/kiro-device-login.ts +312 -0
  144. package/src/oauth/kiro-terminal-failover.ts +28 -0
  145. package/src/oauth/login-flow-state.ts +7 -1
  146. package/src/oauth/pool-settings-capability.ts +23 -14
  147. package/src/oauth/store.ts +56 -3
  148. package/src/oauth/types.ts +4 -0
  149. package/src/plugins/loader.ts +350 -0
  150. package/src/plugins/upstream-hooks.ts +119 -0
  151. package/src/protocols/encoders/adapter-events.ts +9 -2
  152. package/src/providers/account-quota-disk.ts +42 -4
  153. package/src/providers/codebuddy-models.ts +6 -2
  154. package/src/providers/command-code-efforts.ts +33 -10
  155. package/src/providers/kiro-account-state-disk.ts +44 -0
  156. package/src/providers/kiro-model-catalog.ts +162 -0
  157. package/src/providers/kiro-models.ts +5 -4
  158. package/src/providers/kiro-quota-metrics.ts +35 -0
  159. package/src/providers/kiro-usage.ts +118 -14
  160. package/src/providers/quota/account-cache.ts +43 -7
  161. package/src/providers/quota/antigravity.ts +8 -3
  162. package/src/providers/quota/kiro-account-probe.ts +12 -0
  163. package/src/providers/quota/vendor-probes-key.ts +31 -26
  164. package/src/providers/quota/vendor-probes-oauth.ts +16 -7
  165. package/src/providers/quota-types.ts +14 -0
  166. package/src/providers/quota.ts +21 -21
  167. package/src/providers/registry/entries-extended.ts +5 -3
  168. package/src/providers/request-pacing.ts +410 -12
  169. package/src/remote-control/workspace-codex-runtime.ts +4 -3
  170. package/src/responses/citation-markers.ts +132 -65
  171. package/src/responses/hosted-tool-policy.ts +14 -3
  172. package/src/responses/parser-content.ts +3 -2
  173. package/src/responses/parser.ts +4 -1
  174. package/src/responses/schema.ts +4 -0
  175. package/src/responses/visualization-directives.ts +182 -0
  176. package/src/server/chat-native-sse.ts +24 -12
  177. package/src/server/chat-native.ts +137 -68
  178. package/src/server/claude-messages.ts +147 -12
  179. package/src/server/index/link-listener.ts +7 -0
  180. package/src/server/index/optional-listeners.ts +4 -1
  181. package/src/server/index/package-tree-guard.ts +35 -16
  182. package/src/server/index/serve-options.ts +4 -2
  183. package/src/server/index/startup-warnings.ts +6 -0
  184. package/src/server/index.ts +7 -7
  185. package/src/server/inference/client-encoder-delivery.ts +3 -0
  186. package/src/server/inference/context.ts +30 -2
  187. package/src/server/local-account-switch-auth.ts +79 -0
  188. package/src/server/management/agent-settings-routes.ts +11 -3
  189. package/src/server/management/config-routes.ts +28 -11
  190. package/src/server/management/context.ts +2 -0
  191. package/src/server/management/link-routes.ts +123 -39
  192. package/src/server/management/logs-usage-routes.ts +9 -42
  193. package/src/server/management/model-rows.ts +1 -0
  194. package/src/server/management/oauth-account-routes.ts +67 -12
  195. package/src/server/management/route-registry.ts +5 -5
  196. package/src/server/management/sibling-guard.ts +60 -0
  197. package/src/server/management/storage-log-guard-routes.ts +72 -6
  198. package/src/server/management/system-restart.ts +92 -135
  199. package/src/server/management/system-routes.ts +1 -1
  200. package/src/server/management-api.ts +31 -3
  201. package/src/server/management-auth.ts +3 -0
  202. package/src/server/port-reclaim.ts +111 -1
  203. package/src/server/proxy-liveness.ts +16 -0
  204. package/src/server/relay.ts +16 -15
  205. package/src/server/request-log-filter.ts +28 -27
  206. package/src/server/request-log.ts +7 -3
  207. package/src/server/request-metrics.ts +28 -0
  208. package/src/server/responses/adapter-continuation.ts +151 -21
  209. package/src/server/responses/adapter-delivery.ts +21 -7
  210. package/src/server/responses/adapter-dispatch.ts +239 -51
  211. package/src/server/responses/codex-ws-pool.ts +5 -4
  212. package/src/server/responses/codex-ws-request.ts +4 -1
  213. package/src/server/responses/compact.ts +8 -4
  214. package/src/server/responses/compaction-recovery-policy.ts +116 -0
  215. package/src/server/responses/compaction-recovery.ts +330 -0
  216. package/src/server/responses/core-combo-failure.ts +1 -0
  217. package/src/server/responses/core-combo.ts +5 -1
  218. package/src/server/responses/core-lifetime.ts +22 -0
  219. package/src/server/responses/core-options.ts +16 -1
  220. package/src/server/responses/core.ts +14 -4
  221. package/src/server/responses/empty-completion-guard.ts +2 -0
  222. package/src/server/responses/fetch-helpers.ts +82 -31
  223. package/src/server/responses/passthrough-delivery.ts +3 -2
  224. package/src/server/responses/passthrough-dispatch.ts +12 -2
  225. package/src/server/responses/passthrough-execution.ts +2 -2
  226. package/src/server/responses/request-send-budget.ts +21 -5
  227. package/src/server/responses/request-transport.ts +116 -18
  228. package/src/server/responses/reset-replay.ts +32 -0
  229. package/src/server/responses/run-turn-execution.ts +136 -19
  230. package/src/server/responses/sidecar-execution.ts +41 -8
  231. package/src/server/responses/terminal-guard.ts +2 -0
  232. package/src/server/responses/ws-upstream.ts +44 -4
  233. package/src/server/restart-replacement.ts +382 -0
  234. package/src/server/startup-health-cache.ts +23 -4
  235. package/src/server/stop-teardown.ts +35 -1
  236. package/src/server/system-env.ts +3 -0
  237. package/src/service/windows-taskxml.ts +11 -7
  238. package/src/service/windows-wrapper-exit.ts +8 -0
  239. package/src/stall-timeout.ts +35 -10
  240. package/src/storage/policy-job.ts +4 -0
  241. package/src/storage/scanner.ts +177 -29
  242. package/src/storage/storage-mutation-coordinator.ts +17 -0
  243. package/src/tray/windows-tray.ps1 +50 -23
  244. package/src/types/config.ts +4 -0
  245. package/src/types/provider.ts +5 -1
  246. package/src/types/request.ts +13 -2
  247. package/src/update/index.ts +3 -1
  248. package/src/update/job.ts +6 -2
  249. package/src/update/mise-launcher-target.ts +100 -0
  250. package/src/update/worker-launch.ts +46 -0
  251. package/src/usage/log.ts +2 -0
  252. package/src/web-search/loop.ts +5 -2
  253. package/gui/dist/assets/App-BJsT8Icc.css +0 -1
  254. package/gui/dist/assets/App-D3pNGiN4.js +0 -51
  255. package/gui/dist/assets/Tray-yb05wCjb.js +0 -1
  256. package/gui/dist/assets/index-CouvvtMV.js +0 -86
  257. package/gui/dist/assets/tray-data-CYGdjCJ7.js +0 -1
@@ -1,8 +1,34 @@
1
1
  import type { OcxProviderConfig, RequestPacingRule } from "../types";
2
2
  import type { GenerationContext } from "../lib/state-store-sweeper";
3
+ import { carryReplayRefusal, isNonReplayableResponse, markResponseNonReplayable } from "../lib/upstream-retry";
4
+ import { redactSecretString } from "../lib/redact";
3
5
 
4
6
  export const REQUEST_PACING_MAX_QUEUE_DEPTH = 256;
5
7
  export const REQUEST_PACING_MAX_QUEUE_AGE_MS = 60_000;
8
+ /**
9
+ * A leased response body that is neither read nor cancelled for this long releases its
10
+ * lease and cancels the body: a caller that dropped a Response without touching its body
11
+ * must not hold a concurrency slot until process restart. The first pull or cancel swaps
12
+ * this deadline for the longer inactivity window below.
13
+ */
14
+ export const REQUEST_PACING_UNCONSUMED_BODY_MS = 30_000;
15
+ /**
16
+ * Once a tracked body has been pulled at least once, this much silence reclaims its lease
17
+ * and cancels the body: a consumer that reads part of a response and then abandons it
18
+ * without cancel() (a clone-based peek that cancels only its own tee branch) must not
19
+ * hold a concurrency slot forever either. The window is far longer than the unconsumed
20
+ * deadline because a live consumer waiting on a slow upstream is indistinguishable from
21
+ * an abandoned one at this layer; providers keep SSE connections alive with keep-alives
22
+ * well inside this window.
23
+ */
24
+ export const REQUEST_PACING_BODY_INACTIVITY_MS = 300_000;
25
+ /**
26
+ * Retry-After floor for a waiter refused while blocked by a concurrency cap rather than
27
+ * by a spacing interval. Interval readiness computes to zero under a pure cap, and the
28
+ * 1s minimum would invite clients to re-hit a saturated provider every second, which is
29
+ * the shared-account 429 churn this feature exists to prevent.
30
+ */
31
+ export const REQUEST_PACING_CONCURRENCY_RETRY_AFTER_SECONDS = 5;
6
32
 
7
33
  let maxQueueDepth = REQUEST_PACING_MAX_QUEUE_DEPTH;
8
34
  let maxQueueAgeMs = REQUEST_PACING_MAX_QUEUE_AGE_MS;
@@ -33,9 +59,11 @@ interface Waiter {
33
59
  modelId?: string;
34
60
  providerIntervalMs: number;
35
61
  modelIntervalMs: number;
62
+ providerMaxConcurrent: number;
63
+ modelMaxConcurrent: number;
36
64
  queuedAt: number;
37
65
  signal?: AbortSignal;
38
- resolve: () => void;
66
+ resolve: (slot: ProviderRequestSlot) => void;
39
67
  reject: (reason: unknown) => void;
40
68
  abort?: () => void;
41
69
  }
@@ -44,11 +72,51 @@ interface ProviderPacer {
44
72
  queue: Waiter[];
45
73
  providerNextStartAt: number;
46
74
  modelNextStartAt: Map<string, number>;
75
+ providerInFlight: number;
76
+ modelInFlight: Map<string, number>;
47
77
  timer?: unknown;
48
78
  lastStartedAt?: number;
49
79
  lastModelId?: string;
50
80
  }
51
81
 
82
+ /**
83
+ * Lease returned by waitForProviderRequestSlot. Inert unless the provider (or the
84
+ * request model override) sets maxConcurrentRequests; then release() returns the
85
+ * concurrency slot when the upstream request finishes, whether that is body completion,
86
+ * body cancellation, or a send that never produced a response.
87
+ */
88
+ export interface ProviderRequestSlot {
89
+ /** Provider identity for operator diagnostics: names the lease holder in deadline warnings. */
90
+ readonly providerName?: string;
91
+ /**
92
+ * True only when this slot holds a concurrency lease whose release must follow the
93
+ * upstream body lifecycle. Interval-only slots stay inert so response objects keep
94
+ * their identity through the fetch path.
95
+ */
96
+ readonly leased: boolean;
97
+ /**
98
+ * True once a tracked response body owns this lease, meaning body completion, body
99
+ * cancellation, or the unconsumed-body deadline will release it. Turn-end cleanup must
100
+ * then leave the release to that lifecycle instead of returning the lease early.
101
+ */
102
+ readonly bodyTracked: boolean;
103
+ /**
104
+ * True once release() has run. A turn-scoped transport uses this to stop pacing by
105
+ * interval after the lease it relied on is gone (a send that threw, a body that
106
+ * closed) and re-acquire with concurrency, so the cap keeps counting its follow-ups.
107
+ */
108
+ readonly released: boolean;
109
+ /** Idempotent. Safe to call from body completion, cancellation, and error paths alike. */
110
+ release(): void;
111
+ }
112
+
113
+ /** Internal: leased slots expose this to trackProviderRequestSlotBody when a body takes over the release. */
114
+ interface BodyTrackableProviderRequestSlot extends ProviderRequestSlot {
115
+ markBodyTracked(): void;
116
+ }
117
+
118
+ const inertProviderRequestSlot: ProviderRequestSlot = { leased: false, bodyTracked: false, released: false, release() {} };
119
+
52
120
  export interface RequestPacingRuntime {
53
121
  now: () => number;
54
122
  setTimer: (callback: () => void, delayMs: number) => unknown;
@@ -61,6 +129,7 @@ export interface ProviderRequestPacingStatus {
61
129
  enabled: boolean;
62
130
  queued: number;
63
131
  nextSlotInMs: number;
132
+ inFlight?: number;
64
133
  lastStartedAt?: number;
65
134
  lastModelId?: string;
66
135
  }
@@ -90,6 +159,15 @@ function normalizedInterval(rule: RequestPacingRule | undefined): number {
90
159
  return Math.max(rpmInterval, fixedInterval);
91
160
  }
92
161
 
162
+ function normalizedMaxConcurrent(rule: RequestPacingRule | undefined): number {
163
+ return typeof rule?.maxConcurrentRequests === "number" && rule.maxConcurrentRequests > 0
164
+ // Floor so runtime-injected configs that bypass the integer schema cannot admit
165
+ // one request past the configured ceiling (a 2.5 cap must not let a third start);
166
+ // clamp sub-1 fractionals up to 1 so a positive cap never degrades into none.
167
+ ? Math.max(1, Math.floor(rule.maxConcurrentRequests))
168
+ : 0;
169
+ }
170
+
93
171
  export function requestPacingIntervalMs(provider: OcxProviderConfig, modelId?: string): number {
94
172
  const policy = provider.requestPacing;
95
173
  if (!policy?.enabled) return 0;
@@ -100,15 +178,34 @@ export function requestPacingIntervalMs(provider: OcxProviderConfig, modelId?: s
100
178
  function requestPacingIntervals(provider: OcxProviderConfig, modelId?: string): {
101
179
  providerIntervalMs: number;
102
180
  modelIntervalMs: number;
181
+ providerMaxConcurrent: number;
182
+ modelMaxConcurrent: number;
103
183
  } {
104
184
  const policy = provider.requestPacing;
105
- if (!policy?.enabled) return { providerIntervalMs: 0, modelIntervalMs: 0 };
185
+ if (!policy?.enabled) {
186
+ return { providerIntervalMs: 0, modelIntervalMs: 0, providerMaxConcurrent: 0, modelMaxConcurrent: 0 };
187
+ }
106
188
  return {
107
189
  providerIntervalMs: normalizedInterval(policy),
108
190
  modelIntervalMs: modelId ? normalizedInterval(policy.models?.[modelId]) : 0,
191
+ providerMaxConcurrent: normalizedMaxConcurrent(policy),
192
+ modelMaxConcurrent: modelId ? normalizedMaxConcurrent(policy.models?.[modelId]) : 0,
109
193
  };
110
194
  }
111
195
 
196
+ /**
197
+ * Whether a concurrency cap applies to this provider/model, expressed as the looser of
198
+ * the two configured bounds. Enforcement gates the provider cap and the model cap
199
+ * independently (a model override of 1 inside a provider cap of 5 admits one), so the
200
+ * return value is a cap-presence signal — every current caller tests it for > 0 — and
201
+ * NOT an effective ceiling. A caller that needs the true ceiling must compute
202
+ * min(provider, model || Infinity) itself.
203
+ */
204
+ export function requestPacingMaxConcurrentRequests(provider: OcxProviderConfig, modelId?: string): number {
205
+ const limits = requestPacingIntervals(provider, modelId);
206
+ return Math.max(limits.providerMaxConcurrent, limits.modelMaxConcurrent);
207
+ }
208
+
112
209
  function waiterReadyAt(state: ProviderPacer, modelId: string | undefined): number {
113
210
  return Math.max(
114
211
  state.providerNextStartAt,
@@ -116,8 +213,65 @@ function waiterReadyAt(state: ProviderPacer, modelId: string | undefined): numbe
116
213
  );
117
214
  }
118
215
 
119
- function pacingRetryAfterSeconds(state: ProviderPacer, modelId: string | undefined, now: number): number {
120
- return Math.max(1, Math.ceil(Math.max(0, waiterReadyAt(state, modelId) - now) / 1000));
216
+ function pacingRetryAfterSeconds(
217
+ state: ProviderPacer,
218
+ modelId: string | undefined,
219
+ now: number,
220
+ concurrencyCapped = false,
221
+ ): number {
222
+ const intervalSeconds = Math.max(1, Math.ceil(Math.max(0, waiterReadyAt(state, modelId) - now) / 1000));
223
+ // A waiter blocked by an in-flight lease has no interval to report: when the wait ends
224
+ // in refusal, the honest answer is that the cap is saturated, not "retry every second".
225
+ return concurrencyCapped
226
+ ? Math.max(intervalSeconds, REQUEST_PACING_CONCURRENCY_RETRY_AFTER_SECONDS)
227
+ : intervalSeconds;
228
+ }
229
+
230
+ function makeProviderRequestSlot(
231
+ providerName: string,
232
+ state: ProviderPacer,
233
+ waiter: Waiter,
234
+ ): BodyTrackableProviderRequestSlot {
235
+ let released = false;
236
+ let bodyTracked = false;
237
+ const leased = waiter.providerMaxConcurrent > 0 || waiter.modelMaxConcurrent > 0;
238
+ const slot: BodyTrackableProviderRequestSlot = {
239
+ providerName,
240
+ leased,
241
+ get bodyTracked() {
242
+ return bodyTracked;
243
+ },
244
+ get released() {
245
+ return released;
246
+ },
247
+ markBodyTracked() {
248
+ bodyTracked = true;
249
+ },
250
+ release() {
251
+ if (released) return;
252
+ released = true;
253
+ waiter.signal?.removeEventListener("abort", releaseOnAbort);
254
+ if (waiter.providerMaxConcurrent > 0 && state.providerInFlight > 0) state.providerInFlight -= 1;
255
+ if (waiter.modelId && waiter.modelMaxConcurrent > 0) {
256
+ const current = state.modelInFlight.get(waiter.modelId) ?? 0;
257
+ if (current <= 1) state.modelInFlight.delete(waiter.modelId);
258
+ else state.modelInFlight.set(waiter.modelId, current - 1);
259
+ }
260
+ runtime.enqueueMicrotask(() => {
261
+ // A pending wake-up timer makes runQueue defer to it, but the lease that just
262
+ // returned may admit a waiter the timer was never scheduled for, so take over.
263
+ if (state.timer) {
264
+ runtime.clearTimer(state.timer);
265
+ state.timer = undefined;
266
+ }
267
+ runQueue(providerName, state);
268
+ });
269
+ },
270
+ };
271
+ const releaseOnAbort = (): void => slot.release();
272
+ waiter.signal?.addEventListener("abort", releaseOnAbort, { once: true });
273
+ if (waiter.signal?.aborted) slot.release();
274
+ return slot;
121
275
  }
122
276
 
123
277
  function rejectExpiredWaiters(providerName: string, state: ProviderPacer, now: number): void {
@@ -129,7 +283,12 @@ function rejectExpiredWaiters(providerName: string, state: ProviderPacer, now: n
129
283
  waiter.reject(new RequestPacingQueueOverloadError(
130
284
  providerName,
131
285
  "queue_expired",
132
- pacingRetryAfterSeconds(state, waiter.modelId, now),
286
+ pacingRetryAfterSeconds(
287
+ state,
288
+ waiter.modelId,
289
+ now,
290
+ waiter.providerMaxConcurrent > 0 || waiter.modelMaxConcurrent > 0,
291
+ ),
133
292
  ));
134
293
  }
135
294
  }
@@ -162,6 +321,9 @@ function runQueue(providerName: string, state: ProviderPacer): void {
162
321
 
163
322
  const providerReadyAt = Math.max(now, state.providerNextStartAt);
164
323
  const waiterIndex = state.queue.findIndex(waiter => {
324
+ if (waiter.providerMaxConcurrent > 0 && state.providerInFlight >= waiter.providerMaxConcurrent) return false;
325
+ if (waiter.modelId && waiter.modelMaxConcurrent > 0
326
+ && (state.modelInFlight.get(waiter.modelId) ?? 0) >= waiter.modelMaxConcurrent) return false;
165
327
  const modelReadyAt = waiter.modelId ? (state.modelNextStartAt.get(waiter.modelId) ?? 0) : 0;
166
328
  return Math.max(providerReadyAt, modelReadyAt) <= now;
167
329
  });
@@ -171,8 +333,15 @@ function runQueue(providerName: string, state: ProviderPacer): void {
171
333
  const modelReadyAt = waiter.modelId ? (state.modelNextStartAt.get(waiter.modelId) ?? 0) : 0;
172
334
  const readyAt = Math.max(providerReadyAt, modelReadyAt);
173
335
  const expiresAt = waiter.queuedAt + maxQueueAgeMs;
174
- earliestAt = Math.min(earliestAt, readyAt, expiresAt);
336
+ // A waiter whose start time already passed is blocked on an in-flight lease; its release
337
+ // re-runs this queue through a microtask, so the timer only needs its expiry backstop.
338
+ // Scheduling for its readyAt (in the past) would spin the timer on every empty pass.
339
+ earliestAt = Math.min(earliestAt, readyAt <= now ? Number.POSITIVE_INFINITY : readyAt, expiresAt);
175
340
  }
341
+ // Unreachable for a non-empty queue: rejectExpiredWaiters above removed every waiter
342
+ // whose expiresAt passed, so each remaining one contributes a finite value. Kept as a
343
+ // belt-and-braces backstop rather than a case future readers should hunt for.
344
+ if (!Number.isFinite(earliestAt)) return;
176
345
  const delayMs = Math.max(0, earliestAt - now);
177
346
  state.timer = runtime.setTimer(() => {
178
347
  state.timer = undefined;
@@ -190,7 +359,11 @@ function runQueue(providerName: string, state: ProviderPacer): void {
190
359
  if (waiter.modelId && waiter.modelIntervalMs > 0) {
191
360
  state.modelNextStartAt.set(waiter.modelId, startedAt + waiter.modelIntervalMs);
192
361
  }
193
- waiter.resolve();
362
+ if (waiter.providerMaxConcurrent > 0) state.providerInFlight += 1;
363
+ if (waiter.modelId && waiter.modelMaxConcurrent > 0) {
364
+ state.modelInFlight.set(waiter.modelId, (state.modelInFlight.get(waiter.modelId) ?? 0) + 1);
365
+ }
366
+ waiter.resolve(makeProviderRequestSlot(providerName, state, waiter));
194
367
  runtime.enqueueMicrotask(() => runQueue(providerName, state));
195
368
  }
196
369
 
@@ -199,13 +372,28 @@ export async function waitForProviderRequestSlot(
199
372
  provider: OcxProviderConfig,
200
373
  modelId?: string,
201
374
  signal?: AbortSignal,
202
- ): Promise<void> {
375
+ options?: { concurrency?: boolean },
376
+ ): Promise<ProviderRequestSlot> {
203
377
  const intervals = requestPacingIntervals(provider, modelId);
204
- if (Math.max(intervals.providerIntervalMs, intervals.modelIntervalMs) <= 0) return;
378
+ // A turn-scoped transport sends several physical requests per logical turn (Cursor
379
+ // HTTP/1.1 RunSSE plus BidiAppends). One lease covers the turn; follow-up sends pace
380
+ // by interval only, or a follow-up would queue behind the lease its own turn holds.
381
+ const concurrency = options?.concurrency !== false;
382
+ const waiterIntervals = concurrency ? intervals : {
383
+ providerIntervalMs: intervals.providerIntervalMs,
384
+ modelIntervalMs: intervals.modelIntervalMs,
385
+ providerMaxConcurrent: 0,
386
+ modelMaxConcurrent: 0,
387
+ };
388
+ const paced = Math.max(intervals.providerIntervalMs, intervals.modelIntervalMs) > 0
389
+ || waiterIntervals.providerMaxConcurrent > 0
390
+ || waiterIntervals.modelMaxConcurrent > 0;
391
+ if (!paced) return inertProviderRequestSlot;
205
392
  if (signal?.aborted) throw abortReason(signal);
206
393
 
207
394
  const state = pacers.get(providerName) ?? {
208
395
  queue: [], providerNextStartAt: 0, modelNextStartAt: new Map<string, number>(),
396
+ providerInFlight: 0, modelInFlight: new Map<string, number>(),
209
397
  };
210
398
  pacers.set(providerName, state);
211
399
 
@@ -221,12 +409,24 @@ export async function waitForProviderRequestSlot(
221
409
  throw new RequestPacingQueueOverloadError(
222
410
  providerName,
223
411
  "queue_full",
224
- pacingRetryAfterSeconds(state, modelId, runtime.now()),
412
+ pacingRetryAfterSeconds(
413
+ state,
414
+ modelId,
415
+ runtime.now(),
416
+ waiterIntervals.providerMaxConcurrent > 0 || waiterIntervals.modelMaxConcurrent > 0,
417
+ ),
225
418
  );
226
419
  }
227
420
 
228
- await new Promise<void>((resolve, reject) => {
229
- const waiter: Waiter = { modelId, ...intervals, queuedAt: runtime.now(), signal, resolve, reject };
421
+ return await new Promise<ProviderRequestSlot>((resolve, reject) => {
422
+ const waiter: Waiter = {
423
+ modelId,
424
+ ...waiterIntervals,
425
+ queuedAt: runtime.now(),
426
+ signal,
427
+ resolve,
428
+ reject,
429
+ };
230
430
  waiter.abort = () => {
231
431
  const index = state.queue.indexOf(waiter);
232
432
  if (index >= 0) state.queue.splice(index, 1);
@@ -252,6 +452,192 @@ export async function waitForProviderRequestSlot(
252
452
  });
253
453
  }
254
454
 
455
+ /**
456
+ * Tie a pacing slot lease to an upstream response body: the lease is released when the
457
+ * body completes, errors, or is cancelled by the consumer. A null body (204/304/HEAD)
458
+ * means the exchange is already finished, so the lease returns immediately. A body that
459
+ * is neither read nor cancelled for REQUEST_PACING_UNCONSUMED_BODY_MS releases its lease
460
+ * and cancels the body, so a dropped Response cannot hold a slot forever. The returned
461
+ * Response preserves status, statusText, headers, and the identity-based replay markers.
462
+ */
463
+ export function trackProviderRequestSlotBody(
464
+ slot: ProviderRequestSlot | undefined,
465
+ response: Response,
466
+ ): Response {
467
+ // An unleased slot has nothing to return on body close, and rewrapping the Response
468
+ // would break identity-based markers (the eager WS relay registry is a WeakSet).
469
+ if (!slot?.leased) return response;
470
+ // Slots built outside waitForProviderRequestSlot (test doubles) satisfy only the public
471
+ // ProviderRequestSlot shape; their bodies still own the release through the wrapper below.
472
+ (slot as Partial<BodyTrackableProviderRequestSlot>).markBodyTracked?.();
473
+ if (!response.body) {
474
+ slot.release();
475
+ return response;
476
+ }
477
+ const source = response.body;
478
+ let released = false;
479
+ // Set when a deadline callback cancelled the source: the truncation must surface as a
480
+ // stream error on the next pull, never as a clean EOF a relay would treat as success.
481
+ let expired = false;
482
+ let expiryTimer: unknown;
483
+ let expiryKind: "unconsumed" | "inactive" = "unconsumed";
484
+ const clearExpiryTimer = (): void => {
485
+ if (expiryTimer === undefined) return;
486
+ runtime.clearTimer(expiryTimer);
487
+ expiryTimer = undefined;
488
+ };
489
+ const release = (): void => {
490
+ if (released) return;
491
+ released = true;
492
+ clearExpiryTimer();
493
+ slot.release();
494
+ };
495
+ const armExpiryTimer = (delayMs: number, kind: "unconsumed" | "inactive"): void => {
496
+ clearExpiryTimer();
497
+ expiryKind = kind;
498
+ expiryTimer = runtime.setTimer(() => {
499
+ expiryTimer = undefined;
500
+ if (released) return;
501
+ expired = true;
502
+ release();
503
+ // Cancel through the reader when one exists: after the first pull the source is
504
+ // locked to it, and cancelling a locked stream rejects without releasing the
505
+ // socket — the exact upstream leak this deadline exists to prevent.
506
+ void (reader ?? source).cancel().catch(() => {});
507
+ const who = slot.providerName === undefined ? "" : ` for provider ${JSON.stringify(redactSecretString(slot.providerName))}`;
508
+ console.warn(
509
+ expiryKind === "inactive"
510
+ ? `[opencodex] requestPacing${who} released a concurrency lease after `
511
+ + REQUEST_PACING_BODY_INACTIVITY_MS
512
+ + "ms of response body inactivity; the body was cancelled."
513
+ : `[opencodex] requestPacing${who} released a concurrency lease after `
514
+ + REQUEST_PACING_UNCONSUMED_BODY_MS
515
+ + "ms because the provider response body was neither read nor cancelled; the body was cancelled.",
516
+ );
517
+ }, delayMs);
518
+ };
519
+ armExpiryTimer(REQUEST_PACING_UNCONSUMED_BODY_MS, "unconsumed");
520
+ let reader: ReadableStreamDefaultReader<Uint8Array> | undefined;
521
+ let cancelled = false;
522
+ const tracked = new ReadableStream<Uint8Array>({
523
+ pull: async controller => {
524
+ // A pull proves a consumer is attached, not that it will keep reading: re-arm a
525
+ // longer inactivity deadline instead of disarming. A body read once and then
526
+ // abandoned (a clone-based peek that cancels only its own tee branch) must still
527
+ // return its lease, while a live stream keeps pushing the deadline back per pull.
528
+ if (!released) armExpiryTimer(REQUEST_PACING_BODY_INACTIVITY_MS, "inactive");
529
+ try {
530
+ // Inside the try: a source another reader already locked makes getReader()
531
+ // throw, and with the deadline disarmed above that failure must release the
532
+ // lease here instead of stranding it for the process lifetime.
533
+ reader ??= source.getReader();
534
+ const { done, value } = await reader.read();
535
+ // A consumer cancel while this read was pending resolves it (done or a late chunk);
536
+ // touching the cancelled controller would throw from the pull algorithm.
537
+ if (cancelled) return;
538
+ if (expired) {
539
+ // The deadline cancelled the source mid-stream; report the truncation as an
540
+ // error instead of letting the cancelled read's done flag close the stream.
541
+ controller.error(new Error(
542
+ "[opencodex] requestPacing cancelled the response body after its lease deadline",
543
+ ));
544
+ return;
545
+ }
546
+ if (done) {
547
+ controller.close();
548
+ release();
549
+ return;
550
+ }
551
+ controller.enqueue(value);
552
+ } catch (error) {
553
+ if (cancelled) return;
554
+ release();
555
+ controller.error(error);
556
+ }
557
+ },
558
+ cancel: reason => {
559
+ cancelled = true;
560
+ clearExpiryTimer();
561
+ release();
562
+ return (reader ?? source).cancel(reason);
563
+ },
564
+ }, {
565
+ // A default-count stream pulls once at construction with no reader attached, which
566
+ // would mark the body consumed and disarm the deadline before any real consumer
567
+ // arrives. Zero capacity keeps pull consumer-driven: the first read arms nothing
568
+ // until a reader actually asks for bytes.
569
+ highWaterMark: 0,
570
+ });
571
+ let wrapped: Response;
572
+ try {
573
+ wrapped = new Response(tracked, {
574
+ status: response.status,
575
+ statusText: response.statusText,
576
+ headers: response.headers,
577
+ });
578
+ } catch {
579
+ // A non-conforming status (a proxy passing a raw 6xx through) or a body on a
580
+ // null-body status throws here after markBodyTracked, and boundary cleanup skips
581
+ // body-tracked slots: return the lease now instead of waiting out the deadline.
582
+ // The abandoned rewrap never locked the source (no pull ran), so hand back the
583
+ // ORIGINAL response with its body intact: rethrowing would make the google-http,
584
+ // command-code and mimo retry ladders replay a request whose response did arrive.
585
+ release();
586
+ return response;
587
+ }
588
+ // Retry helpers mark the response they return from, and recovery decisions key on these
589
+ // identity markers; the rewrap must not make a non-replayable response look replayable.
590
+ if (isNonReplayableResponse(response)) markResponseNonReplayable(wrapped);
591
+ return carryReplayRefusal(response, wrapped);
592
+ }
593
+
594
+ /**
595
+ * Return a lease at a turn or attempt boundary: a lease a tracked response body now owns
596
+ * is left to that body's lifecycle, and any other unconsumed lease is released. Idempotent,
597
+ * and a no-op for interval-only slots, so every boundary can call it unconditionally.
598
+ */
599
+ export function releaseProviderRequestSlot(slot: ProviderRequestSlot | undefined): void {
600
+ if (!slot || slot.bodyTracked) return;
601
+ slot.release();
602
+ }
603
+
604
+ /** Transfer a physical send's lease to its response body, or release it on failure. */
605
+ export async function sendTrackingRequestSlot(
606
+ slot: ProviderRequestSlot | undefined,
607
+ send: () => Promise<Response>,
608
+ ): Promise<Response> {
609
+ try {
610
+ return trackProviderRequestSlotBody(slot, await send());
611
+ } catch (error) {
612
+ slot?.release();
613
+ throw error;
614
+ }
615
+ }
616
+
617
+ /**
618
+ * Acquire one pacing lease and return it at the boundary. The send callback runs with
619
+ * the lease; this helper releases it unless a tracked response body has taken ownership.
620
+ * A send that throws (an abort, a send-budget refusal, a build failure) or returns
621
+ * without ever dispatching through the executor would otherwise strand the lease for
622
+ * the process lifetime, and a dispatched send's tracked body keeps its own release.
623
+ * Centralized so every dispatch boundary shares one copy of the acquire/release pairing
624
+ * and a future edit cannot fork the lease lifecycle.
625
+ */
626
+ export async function withProviderRequestSlot<T>(
627
+ providerName: string,
628
+ provider: OcxProviderConfig,
629
+ modelId: string | undefined,
630
+ signal: AbortSignal | undefined,
631
+ send: (pacingSlot: ProviderRequestSlot) => Promise<T>,
632
+ ): Promise<T> {
633
+ const pacingSlot = await waitForProviderRequestSlot(providerName, provider, modelId, signal);
634
+ try {
635
+ return await send(pacingSlot);
636
+ } finally {
637
+ releaseProviderRequestSlot(pacingSlot);
638
+ }
639
+ }
640
+
255
641
  export function providerRequestPacingStatus(
256
642
  providerName: string,
257
643
  provider: OcxProviderConfig,
@@ -266,11 +652,23 @@ export function providerRequestPacingStatus(
266
652
  }
267
653
  if (Number.isFinite(earliestQueuedSlotAt)) nextSlotAt = earliestQueuedSlotAt;
268
654
  }
655
+ const providerConcurrencyCap = requestPacingMaxConcurrentRequests(provider);
656
+ // Gate on enabled so the status surface agrees with enforcement: a disabled requestPacing
657
+ // block must not report in-flight leases alongside enabled: false.
658
+ const anyConcurrencyCap = provider.requestPacing?.enabled === true && (providerConcurrencyCap > 0
659
+ || Object.values(provider.requestPacing?.models ?? {}).some(rule => normalizedMaxConcurrent(rule) > 0));
269
660
  return {
270
661
  provider: providerName,
271
662
  enabled: provider.requestPacing?.enabled === true,
272
663
  queued: state?.queue.length ?? 0,
273
664
  nextSlotInMs: Math.max(0, Math.ceil(nextSlotAt - now)),
665
+ // A provider cap counts every paced send in providerInFlight; model-only caps keep
666
+ // providerInFlight at zero, so report the per-model sum for those providers instead.
667
+ ...(anyConcurrencyCap ? {
668
+ inFlight: providerConcurrencyCap > 0
669
+ ? state?.providerInFlight ?? 0
670
+ : [...(state?.modelInFlight.values() ?? [])].reduce((sum, count) => sum + count, 0),
671
+ } : {}),
274
672
  ...(state?.lastStartedAt !== undefined ? { lastStartedAt: state.lastStartedAt } : {}),
275
673
  ...(state?.lastModelId ? { lastModelId: state.lastModelId } : {}),
276
674
  };
@@ -1,7 +1,7 @@
1
1
  import { chmodSync, linkSync, mkdirSync, mkdtempSync, realpathSync, symlinkSync } from "node:fs";
2
2
  import { tmpdir } from "node:os";
3
3
  import { dirname, isAbsolute, join } from "node:path";
4
- import { resolveCodexRuntime } from "../codex/runtime";
4
+ import { resolveCodexRuntimeAsync } from "../codex/runtime";
5
5
  import { remoteWorkspaceThreadStartParams } from "./workspace-coordinator";
6
6
  import { startRemoteWorkspaceToolBridge } from "./workspace-tool-bridge";
7
7
  import { truncateRemoteWorkspaceUtf8 } from "./workspace-utf8";
@@ -299,7 +299,8 @@ export class CodexRemoteWorkspaceRuntimeFactory implements RemoteWorkspaceRuntim
299
299
  if (this.options.command && this.options.command.length > 0) {
300
300
  return { available: true, version: this.options.version ?? "test" };
301
301
  }
302
- const resolved = resolveCodexRuntime();
302
+ // Async probes: a Hub availability check must not freeze every other proxy request.
303
+ const resolved = await resolveCodexRuntimeAsync();
303
304
  const compatibility = codexRemotePermissionProfileCompatibility();
304
305
  if (!compatibility.compatible) return { available: false, reason: compatibility.reason };
305
306
  return resolved.runtime.version
@@ -310,7 +311,7 @@ export class CodexRemoteWorkspaceRuntimeFactory implements RemoteWorkspaceRuntim
310
311
  async start(options: Parameters<RemoteWorkspaceRuntimeFactory["start"]>[0]): Promise<RemoteWorkspaceRuntimeHandle> {
311
312
  const command = this.options.command
312
313
  ? [...this.options.command]
313
- : [resolveCodexRuntime().runtime.command];
314
+ : [(await resolveCodexRuntimeAsync()).runtime.command];
314
315
  if (command.length < 1) throw new Error("Codex CLI is unavailable on this Hub");
315
316
  const executablePath = isAbsolute(command[0]!) ? command[0]! : findExecutableOnPath(command[0]!);
316
317
  if (!executablePath) throw new Error("Codex CLI executable could not be resolved on this Hub");