@bitkyc08/opencodex 2.67.0-preview.20260926 → 2.68.0-preview.20260927

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (257) hide show
  1. package/bin/ocx.mjs +3 -0
  2. package/gui/dist/assets/App-B5-cAEd2.js +51 -0
  3. package/gui/dist/assets/App-DCBismRi.css +1 -0
  4. package/gui/dist/assets/Tray-DFnMiyD0.js +1 -0
  5. package/gui/dist/assets/{index-DtNmX7hW.css → index-BDUBS8PW.css} +1 -1
  6. package/gui/dist/assets/index-BcXblvem.js +86 -0
  7. package/gui/dist/assets/quota-summary-i84eOU3o.js +1 -0
  8. package/gui/dist/index.html +21 -2
  9. package/package.json +1 -1
  10. package/src/adapters/anthropic-image-guard.ts +13 -1
  11. package/src/adapters/anthropic-image-normalize.ts +3 -2
  12. package/src/adapters/anthropic-output-schema.ts +36 -0
  13. package/src/adapters/anthropic.ts +10 -2
  14. package/src/adapters/base.ts +7 -0
  15. package/src/adapters/codebuddy/live-models.ts +180 -0
  16. package/src/adapters/coding-agent/protocol.ts +91 -12
  17. package/src/adapters/coding-agent/turn.ts +32 -27
  18. package/src/adapters/devin/cloud-direct/index.ts +1 -0
  19. package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +18 -4
  20. package/src/adapters/devin.ts +9 -6
  21. package/src/adapters/google-http.ts +4 -4
  22. package/src/adapters/google.ts +93 -3
  23. package/src/adapters/kiro/adapter.ts +6 -2
  24. package/src/adapters/kiro/stream.ts +12 -2
  25. package/src/adapters/kiro/usage.ts +3 -1
  26. package/src/adapters/kiro-errors.ts +11 -1
  27. package/src/adapters/kiro-events.ts +27 -1
  28. package/src/adapters/kiro-refusal.ts +30 -0
  29. package/src/adapters/kiro-retry.ts +71 -31
  30. package/src/adapters/openai-chat/deepseek-artifact-schema.ts +45 -0
  31. package/src/adapters/openai-chat/messages.ts +31 -4
  32. package/src/adapters/openai-chat/serialized-tool-call-content.ts +52 -7
  33. package/src/adapters/openai-chat/tool-call-id-remint.ts +65 -0
  34. package/src/adapters/openai-chat/tool-schema.ts +6 -2
  35. package/src/adapters/openai-chat.ts +2 -2
  36. package/src/adapters/openai-responses/muse-tool-choice.ts +31 -0
  37. package/src/adapters/openai-responses/passthrough.ts +9 -3
  38. package/src/adapters/opencode-go-additional-tools.ts +19 -10
  39. package/src/adapters/physical-send.ts +10 -3
  40. package/src/adapters/registry.ts +2 -1
  41. package/src/adapters/responses-tool-schema.ts +31 -3
  42. package/src/adapters/run-turn-queue.ts +63 -23
  43. package/src/adapters/unique-tool-call-ids.ts +63 -0
  44. package/src/adapters/xai-web-search.ts +32 -2
  45. package/src/bridge/sse.ts +14 -2
  46. package/src/chat/inbound.ts +20 -6
  47. package/src/claude/agents-inject.ts +4 -0
  48. package/src/claude/desktop-3p.ts +53 -13
  49. package/src/claude/desktop-profile.ts +41 -4
  50. package/src/claude/inbound-model-options.ts +14 -2
  51. package/src/claude/inbound.ts +1 -1
  52. package/src/claude/intercept/connect-proxy.ts +17 -1
  53. package/src/claude/intercept/local-ca.ts +7 -1
  54. package/src/claude/message-threads.ts +28 -0
  55. package/src/cli/account-api.ts +13 -0
  56. package/src/cli/account-auth.ts +49 -2
  57. package/src/cli/account-extended.ts +4 -2
  58. package/src/cli/account.ts +4 -2
  59. package/src/cli/capabilities.ts +17 -0
  60. package/src/cli/claude.ts +3 -1
  61. package/src/cli/dispatch.ts +62 -5
  62. package/src/cli/index.ts +84 -79
  63. package/src/cli/minimax.ts +4 -2
  64. package/src/cli/opencode.ts +6 -1
  65. package/src/cli/restart-handoff.ts +110 -0
  66. package/src/cli/status.ts +51 -0
  67. package/src/client/connect.ts +34 -15
  68. package/src/client/link-ingress.ts +102 -0
  69. package/src/client/link-join.ts +37 -16
  70. package/src/client/link-relay.ts +228 -45
  71. package/src/client/link-state.ts +54 -0
  72. package/src/client/link-status.ts +39 -0
  73. package/src/client/link-teardown.ts +2 -2
  74. package/src/client/link-tunnel.ts +615 -86
  75. package/src/client/machine-api.ts +10 -4
  76. package/src/client/machine-listener.ts +57 -10
  77. package/src/client/runtime.ts +210 -35
  78. package/src/clients/config-export/constants.ts +1 -12
  79. package/src/clients/config-export/contracts.ts +2 -0
  80. package/src/clients/config-export/model-metadata.ts +9 -3
  81. package/src/clients/config-export/omp.ts +1 -1
  82. package/src/clients/config-export/zcode-store.ts +2 -2
  83. package/src/clients/config-export.ts +36 -3
  84. package/src/codex/app-server-processes.ts +32 -0
  85. package/src/codex/app-server-restart-service.ts +20 -3
  86. package/src/codex/catalog/access-programs.ts +50 -0
  87. package/src/codex/catalog/build-entries.ts +13 -0
  88. package/src/codex/catalog/derive-entry.ts +1 -0
  89. package/src/codex/catalog/model-hints.ts +11 -0
  90. package/src/codex/catalog/parsing.ts +6 -2
  91. package/src/codex/catalog/provider-models.ts +81 -0
  92. package/src/codex/catalog/retained-sync.ts +3 -0
  93. package/src/codex/codex-write-lock.ts +2 -2
  94. package/src/codex/convergence.ts +2 -0
  95. package/src/codex/desired-state.ts +14 -4
  96. package/src/codex/home.ts +21 -3
  97. package/src/codex/inject/restore.ts +17 -0
  98. package/src/codex/inject/routing-classify.ts +3 -2
  99. package/src/codex/inject-coordination.ts +4 -1
  100. package/src/codex/inject.ts +8 -1
  101. package/src/codex/journal.ts +6 -2
  102. package/src/codex/management-convergence.ts +9 -0
  103. package/src/codex/model-entitlements.ts +28 -5
  104. package/src/codex/routing/idle-window.ts +58 -0
  105. package/src/codex/routing-drift.ts +134 -0
  106. package/src/codex/routing-healer.ts +419 -0
  107. package/src/codex/routing.ts +10 -0
  108. package/src/codex/runtime.ts +188 -57
  109. package/src/codex/sibling-handoff.ts +71 -0
  110. package/src/codex/sibling-start.ts +139 -0
  111. package/src/codex/sync.ts +5 -3
  112. package/src/combos/failover.ts +32 -5
  113. package/src/combos/request.ts +11 -3
  114. package/src/combos/reset-window.ts +10 -5
  115. package/src/combos/resolve.ts +36 -6
  116. package/src/config/diagnostics.ts +2 -1
  117. package/src/config/load-degrade.ts +2 -0
  118. package/src/config/paths.ts +7 -1
  119. package/src/config/process-state.ts +10 -1
  120. package/src/config/schema/compaction-recovery.ts +15 -0
  121. package/src/config/schema/config-schema.ts +2 -0
  122. package/src/config/schema/leaf-validators.ts +9 -3
  123. package/src/generated/compatibility-version.json +360 -204
  124. package/src/images/loop.ts +47 -32
  125. package/src/integrations/catalog-refresh.ts +4 -0
  126. package/src/integrations/omp-yaml-source.ts +1 -0
  127. package/src/lib/claude-request-projection.ts +101 -0
  128. package/src/lib/local-account-switch-capability.ts +48 -0
  129. package/src/lib/local-upstream.ts +146 -0
  130. package/src/lib/package-tree-integrity.ts +25 -3
  131. package/src/lib/package-tree-retarget.ts +130 -0
  132. package/src/lib/process-control.ts +4 -1
  133. package/src/lib/system-restart-contract.ts +106 -0
  134. package/src/lib/upstream-retry.ts +57 -1
  135. package/src/link/ports.ts +9 -0
  136. package/src/link/ssh-argv.ts +16 -0
  137. package/src/link/ssh-runner.ts +70 -2
  138. package/src/link/tunnel-state.ts +58 -9
  139. package/src/oauth/account-quota-rank.ts +29 -18
  140. package/src/oauth/generic-account-failover.ts +179 -22
  141. package/src/oauth/index.ts +17 -4
  142. package/src/oauth/kiro-account-load.ts +107 -0
  143. package/src/oauth/kiro-device-login.ts +312 -0
  144. package/src/oauth/kiro-terminal-failover.ts +28 -0
  145. package/src/oauth/login-flow-state.ts +7 -1
  146. package/src/oauth/pool-settings-capability.ts +23 -14
  147. package/src/oauth/store.ts +56 -3
  148. package/src/oauth/types.ts +4 -0
  149. package/src/plugins/loader.ts +350 -0
  150. package/src/plugins/upstream-hooks.ts +119 -0
  151. package/src/protocols/encoders/adapter-events.ts +9 -2
  152. package/src/providers/account-quota-disk.ts +42 -4
  153. package/src/providers/codebuddy-models.ts +6 -2
  154. package/src/providers/command-code-efforts.ts +33 -10
  155. package/src/providers/kiro-account-state-disk.ts +44 -0
  156. package/src/providers/kiro-model-catalog.ts +162 -0
  157. package/src/providers/kiro-models.ts +5 -4
  158. package/src/providers/kiro-quota-metrics.ts +35 -0
  159. package/src/providers/kiro-usage.ts +118 -14
  160. package/src/providers/quota/account-cache.ts +43 -7
  161. package/src/providers/quota/antigravity.ts +8 -3
  162. package/src/providers/quota/kiro-account-probe.ts +12 -0
  163. package/src/providers/quota/vendor-probes-key.ts +31 -26
  164. package/src/providers/quota/vendor-probes-oauth.ts +16 -7
  165. package/src/providers/quota-types.ts +14 -0
  166. package/src/providers/quota.ts +21 -21
  167. package/src/providers/registry/entries-extended.ts +5 -3
  168. package/src/providers/request-pacing.ts +410 -12
  169. package/src/remote-control/workspace-codex-runtime.ts +4 -3
  170. package/src/responses/citation-markers.ts +132 -65
  171. package/src/responses/hosted-tool-policy.ts +14 -3
  172. package/src/responses/parser-content.ts +3 -2
  173. package/src/responses/parser.ts +4 -1
  174. package/src/responses/schema.ts +4 -0
  175. package/src/responses/visualization-directives.ts +182 -0
  176. package/src/server/chat-native-sse.ts +24 -12
  177. package/src/server/chat-native.ts +137 -68
  178. package/src/server/claude-messages.ts +147 -12
  179. package/src/server/index/link-listener.ts +7 -0
  180. package/src/server/index/optional-listeners.ts +4 -1
  181. package/src/server/index/package-tree-guard.ts +35 -16
  182. package/src/server/index/serve-options.ts +4 -2
  183. package/src/server/index/startup-warnings.ts +6 -0
  184. package/src/server/index.ts +7 -7
  185. package/src/server/inference/client-encoder-delivery.ts +3 -0
  186. package/src/server/inference/context.ts +30 -2
  187. package/src/server/local-account-switch-auth.ts +79 -0
  188. package/src/server/management/agent-settings-routes.ts +11 -3
  189. package/src/server/management/config-routes.ts +28 -11
  190. package/src/server/management/context.ts +2 -0
  191. package/src/server/management/link-routes.ts +123 -39
  192. package/src/server/management/logs-usage-routes.ts +9 -42
  193. package/src/server/management/model-rows.ts +1 -0
  194. package/src/server/management/oauth-account-routes.ts +67 -12
  195. package/src/server/management/route-registry.ts +5 -5
  196. package/src/server/management/sibling-guard.ts +60 -0
  197. package/src/server/management/storage-log-guard-routes.ts +72 -6
  198. package/src/server/management/system-restart.ts +92 -135
  199. package/src/server/management/system-routes.ts +1 -1
  200. package/src/server/management-api.ts +31 -3
  201. package/src/server/management-auth.ts +3 -0
  202. package/src/server/port-reclaim.ts +111 -1
  203. package/src/server/proxy-liveness.ts +16 -0
  204. package/src/server/relay.ts +16 -15
  205. package/src/server/request-log-filter.ts +28 -27
  206. package/src/server/request-log.ts +7 -3
  207. package/src/server/request-metrics.ts +28 -0
  208. package/src/server/responses/adapter-continuation.ts +151 -21
  209. package/src/server/responses/adapter-delivery.ts +21 -7
  210. package/src/server/responses/adapter-dispatch.ts +239 -51
  211. package/src/server/responses/codex-ws-pool.ts +5 -4
  212. package/src/server/responses/codex-ws-request.ts +4 -1
  213. package/src/server/responses/compact.ts +8 -4
  214. package/src/server/responses/compaction-recovery-policy.ts +116 -0
  215. package/src/server/responses/compaction-recovery.ts +330 -0
  216. package/src/server/responses/core-combo-failure.ts +1 -0
  217. package/src/server/responses/core-combo.ts +5 -1
  218. package/src/server/responses/core-lifetime.ts +22 -0
  219. package/src/server/responses/core-options.ts +16 -1
  220. package/src/server/responses/core.ts +14 -4
  221. package/src/server/responses/empty-completion-guard.ts +2 -0
  222. package/src/server/responses/fetch-helpers.ts +82 -31
  223. package/src/server/responses/passthrough-delivery.ts +3 -2
  224. package/src/server/responses/passthrough-dispatch.ts +12 -2
  225. package/src/server/responses/passthrough-execution.ts +2 -2
  226. package/src/server/responses/request-send-budget.ts +21 -5
  227. package/src/server/responses/request-transport.ts +116 -18
  228. package/src/server/responses/reset-replay.ts +32 -0
  229. package/src/server/responses/run-turn-execution.ts +136 -19
  230. package/src/server/responses/sidecar-execution.ts +41 -8
  231. package/src/server/responses/terminal-guard.ts +2 -0
  232. package/src/server/responses/ws-upstream.ts +44 -4
  233. package/src/server/restart-replacement.ts +382 -0
  234. package/src/server/startup-health-cache.ts +23 -4
  235. package/src/server/stop-teardown.ts +35 -1
  236. package/src/server/system-env.ts +3 -0
  237. package/src/service/windows-taskxml.ts +11 -7
  238. package/src/service/windows-wrapper-exit.ts +8 -0
  239. package/src/stall-timeout.ts +35 -10
  240. package/src/storage/policy-job.ts +4 -0
  241. package/src/storage/scanner.ts +177 -29
  242. package/src/storage/storage-mutation-coordinator.ts +17 -0
  243. package/src/tray/windows-tray.ps1 +50 -23
  244. package/src/types/config.ts +4 -0
  245. package/src/types/provider.ts +5 -1
  246. package/src/types/request.ts +13 -2
  247. package/src/update/index.ts +3 -1
  248. package/src/update/job.ts +6 -2
  249. package/src/update/mise-launcher-target.ts +100 -0
  250. package/src/update/worker-launch.ts +46 -0
  251. package/src/usage/log.ts +2 -0
  252. package/src/web-search/loop.ts +5 -2
  253. package/gui/dist/assets/App-B3VRRYSj.js +0 -51
  254. package/gui/dist/assets/App-BJsT8Icc.css +0 -1
  255. package/gui/dist/assets/Tray-TYAA0EQY.js +0 -1
  256. package/gui/dist/assets/index-9Ju7nHsu.js +0 -86
  257. package/gui/dist/assets/tray-data-xKfS2gSG.js +0 -1
@@ -38,11 +38,14 @@ import {
38
38
  UpstreamRetryEvidenceError,
39
39
  type UpstreamSendRecovery,
40
40
  UPSTREAM_RESET_REPLAY_REFUSED_CODE,
41
+ wrapWithZeroOutputRefetch,
41
42
  } from "../lib/upstream-retry";
42
43
  import {
43
44
  isTranslatorBudgetExceededError,
44
45
  type TranslatorBudget,
45
46
  } from "../lib/translator-budget";
47
+ import { authorizeResendForRecovery } from "../lib/request-resend-gate";
48
+ import { ambiguousResendAllowanceFor, selfContainedChatBody } from "./responses/reset-replay";
46
49
  import {
47
50
  hasKeyPoolFailover,
48
51
  selectProactiveApiKeyTransport,
@@ -59,6 +62,7 @@ import type { OcxProviderTransport } from "../providers/xai-transport";
59
62
  import type { RouteResult } from "../router";
60
63
  import type { OcxConfig, OcxProviderConfig } from "../types";
61
64
  import { fetchWithHeaderTimeout, providerFetch, safeHostLabel, sendWithConnectionPolicy } from "./responses/fetch-helpers";
65
+ import { isLocalUpstream } from "../lib/local-upstream";
62
66
  import { linkAbortSignal } from "./responses";
63
67
  import {
64
68
  noteProviderAttemptSend,
@@ -368,7 +372,37 @@ export async function runNativeChatAttempt(
368
372
  );
369
373
  const transientSendAvailable = (): boolean => remainingTransientSends() > 0;
370
374
 
371
- const send = async (request: AdapterRequest, recovery?: "rate-limit-429" | "key-429"): Promise<Response> => {
375
+ /**
376
+ * The operator's replacement grant for THIS logical request, read per leg because
377
+ * `activeProvider` is reassigned by credential rotation inside the send loop.
378
+ *
379
+ * A combo child already holds the request's send ledger, so the grant comes from it and every
380
+ * ambiguous stage of the request draws on the same counter. A direct Chat request opens no such
381
+ * ledger -- it accounts through a spend tracker instead -- so it keeps the single grant locally,
382
+ * which is the same ceiling for the same one attempt.
383
+ */
384
+ const requestIsSelfContained = (() => {
385
+ let memo: boolean | undefined;
386
+ return (): boolean => memo ??= selfContainedChatBody(execution.chatBody);
387
+ })();
388
+ let localAmbiguousResendSpent = false;
389
+ const claimAmbiguousResend = (limit: number): boolean => {
390
+ if (sendBudget?.claimAmbiguousResend) return sendBudget.claimAmbiguousResend(limit);
391
+ if (localAmbiguousResendSpent || limit <= 0) return false;
392
+ localAmbiguousResendSpent = true;
393
+ return true;
394
+ };
395
+ const ambiguousResend = () => ambiguousResendAllowanceFor(
396
+ activeProvider,
397
+ requestIsSelfContained,
398
+ claimAmbiguousResend,
399
+ );
400
+
401
+ const send = async (
402
+ request: AdapterRequest,
403
+ recovery?: "rate-limit-429" | "key-429" | UpstreamSendRecovery,
404
+ singleShot = false,
405
+ ): Promise<Response> => {
372
406
  try {
373
407
  // #2643: opted-in key-auth openai-chat providers retry pre-stream transient statuses on
374
408
  // the native chat lane too; everyone else keeps reset-only semantics.
@@ -377,74 +411,81 @@ export async function runNativeChatAttempt(
377
411
  if (requestTransientPolicy && remaining <= 0) {
378
412
  throw new Error("native Chat transient send budget exhausted before recovery dispatch");
379
413
  }
414
+ const dispatch = (transportRecovery?: UpstreamSendRecovery) => {
415
+ const effectiveRecovery = transportRecovery
416
+ ?? (recovery === "connection-reset" ? "connection-reset" : undefined);
417
+ return fetchWithHeaderTimeout(
418
+ request.url,
419
+ applyUpstreamRecoveryInit({
420
+ method: request.method,
421
+ headers: request.headers,
422
+ body: request.body,
423
+ }, effectiveRecovery),
424
+ upstream.signal,
425
+ connectMs,
426
+ requestedStream,
427
+ providerFetch(activeProvider, undefined, {
428
+ providerName: route.providerName,
429
+ modelId: route.modelId,
430
+ dispatchOverride: async (_input, init, execute) => {
431
+ if (!providerApiKeySelectionIsCurrent(config, route.providerName, activeProvider)) {
432
+ const current = resolveCurrentProviderApiKeyTransport(config, route.providerName, activeProvider);
433
+ if (!current || !isNativeChatRouteEligible({ ...route, provider: current }, execution.chatBody, config)) {
434
+ throw new Error("Provider key selection is no longer available for native Chat");
435
+ }
436
+ activeProvider = current;
437
+ stampApiKeyAccountLabel(logCtx, route.providerName, activeProvider);
438
+ activeAdapter = createOpenAIChatAdapter(current);
439
+ activeRequest.releaseBodyObservation?.();
440
+ releaseRetainedRequest();
441
+ activeRequest = buildActiveRequest();
442
+ try { retainRequest(activeRequest); }
443
+ catch (error) { activeRequest.releaseBodyObservation?.(); throw error; }
444
+ }
445
+ // The retry closure may still hold a pre-pacing request. Replace its entire
446
+ // wire shape, not just Authorization, and retain transport recovery flags.
447
+ request = activeRequest;
448
+ const headers = new Headers(request.headers);
449
+ const encoding = new Headers(init.headers).get("accept-encoding");
450
+ if (!headers.has("accept-encoding") && encoding) headers.set("accept-encoding", encoding);
451
+ if (init.signal?.aborted) throw init.signal.reason;
452
+ if (sendBudget) {
453
+ // Backstop for sends the helper cannot see coming (a reset replay). The first
454
+ // report settles the combo's booking; each later one is charged and booked.
455
+ if (physicalSends > 0 && sendBudget.remainingBaseSends(sharedSendCap) <= 0) {
456
+ throw new SendBudgetExhaustedError(safeHostLabel(request.url));
457
+ }
458
+ physicalSends += 1;
459
+ sendBudget.used += 1;
460
+ } else if (!spendTracker?.charge()) throw new NativeChatSpendRefusal();
461
+ noteProviderAttemptSend(logCtx, route.providerName, activeProvider, logCtx.usageLogInputTokens, transportRecovery ?? recovery);
462
+ // A reselected provider transport is still a physical send: the connection policy
463
+ // and manual-redirect ownership wrap the selected implementation (#4992).
464
+ const dispatched = await sendWithConnectionPolicy(
465
+ (activeProvider as OcxProviderTransport).fetch ?? execute,
466
+ request.url,
467
+ applyUpstreamRecoveryInit({
468
+ ...init, method: request.method, headers, body: request.body,
469
+ }, transportRecovery),
470
+ // Reselection can replace the provider transport and the wire shape, so the
471
+ // egress route is bound to the provider this send actually uses. Omitting it
472
+ // here would let a provider transport bypass its configured route entirely,
473
+ // because that transport wins over the executor that carries the binding.
474
+ { providerName: route.providerName, provider: activeProvider },
475
+ );
476
+ if (!dispatched.ok) await recordKeyAttemptFailure(logCtx, dispatched, init.signal ?? upstream.signal);
477
+ return dispatched;
478
+ },
479
+ }),
480
+ );
481
+ };
482
+ // A replacement bought by an ambiguous reset is ONE send. Routing it back through a retry
483
+ // helper would let that helper spend sends the operator allowance never granted, which is
484
+ // the duplicated inference the allowance exists to bound.
485
+ if (singleShot) return await dispatch("connection-reset");
380
486
  const fetchWithPolicy = requestTransientPolicy ? fetchWithTransientRetry : fetchWithResetRetry;
381
487
  return await fetchWithPolicy(
382
- (transportRecovery?: UpstreamSendRecovery) => {
383
- return fetchWithHeaderTimeout(
384
- request.url,
385
- applyUpstreamRecoveryInit({
386
- method: request.method,
387
- headers: request.headers,
388
- body: request.body,
389
- }, transportRecovery),
390
- upstream.signal,
391
- connectMs,
392
- requestedStream,
393
- providerFetch(activeProvider, undefined, {
394
- providerName: route.providerName,
395
- modelId: route.modelId,
396
- dispatchOverride: async (_input, init, execute) => {
397
- if (!providerApiKeySelectionIsCurrent(config, route.providerName, activeProvider)) {
398
- const current = resolveCurrentProviderApiKeyTransport(config, route.providerName, activeProvider);
399
- if (!current || !isNativeChatRouteEligible({ ...route, provider: current }, execution.chatBody, config)) {
400
- throw new Error("Provider key selection is no longer available for native Chat");
401
- }
402
- activeProvider = current;
403
- stampApiKeyAccountLabel(logCtx, route.providerName, activeProvider);
404
- activeAdapter = createOpenAIChatAdapter(current);
405
- activeRequest.releaseBodyObservation?.();
406
- releaseRetainedRequest();
407
- activeRequest = buildActiveRequest();
408
- try { retainRequest(activeRequest); }
409
- catch (error) { activeRequest.releaseBodyObservation?.(); throw error; }
410
- }
411
- // The retry closure may still hold a pre-pacing request. Replace its entire
412
- // wire shape, not just Authorization, and retain transport recovery flags.
413
- request = activeRequest;
414
- const headers = new Headers(request.headers);
415
- const encoding = new Headers(init.headers).get("accept-encoding");
416
- if (!headers.has("accept-encoding") && encoding) headers.set("accept-encoding", encoding);
417
- if (init.signal?.aborted) throw init.signal.reason;
418
- if (sendBudget) {
419
- // Backstop for sends the helper cannot see coming (a reset replay). The first
420
- // report settles the combo's booking; each later one is charged and booked.
421
- if (physicalSends > 0 && sendBudget.remainingBaseSends(sharedSendCap) <= 0) {
422
- throw new SendBudgetExhaustedError(safeHostLabel(request.url));
423
- }
424
- physicalSends += 1;
425
- sendBudget.used += 1;
426
- } else if (!spendTracker?.charge()) throw new NativeChatSpendRefusal();
427
- noteProviderAttemptSend(logCtx, route.providerName, activeProvider, logCtx.usageLogInputTokens, transportRecovery ?? recovery);
428
- // A reselected provider transport is still a physical send: the connection policy
429
- // and manual-redirect ownership wrap the selected implementation (#4992).
430
- const dispatched = await sendWithConnectionPolicy(
431
- (activeProvider as OcxProviderTransport).fetch ?? execute,
432
- request.url,
433
- applyUpstreamRecoveryInit({
434
- ...init, method: request.method, headers, body: request.body,
435
- }, transportRecovery),
436
- // Reselection can replace the provider transport and the wire shape, so the
437
- // egress route is bound to the provider this send actually uses. Omitting it
438
- // here would let a provider transport bypass its configured route entirely,
439
- // because that transport wins over the executor that carries the binding.
440
- { providerName: route.providerName, provider: activeProvider },
441
- );
442
- if (!dispatched.ok) await recordKeyAttemptFailure(logCtx, dispatched, init.signal ?? upstream.signal);
443
- return dispatched;
444
- },
445
- }),
446
- );
447
- },
488
+ dispatch,
448
489
  {
449
490
  abortSignal: upstream.signal,
450
491
  label: safeHostLabel(request.url),
@@ -629,11 +670,39 @@ export async function runNativeChatAttempt(
629
670
  if (contentType.includes("text/event-stream") && response.body) {
630
671
  if (requestedStream) transferTurnToStream();
631
672
  let terminalStatus: number | undefined;
632
- const stream = nativeChatSse(response.body, {
673
+ const resilientBody = wrapWithZeroOutputRefetch(
674
+ response.body,
675
+ // One physical send, not one more trip through a retry helper: the allowance below buys a
676
+ // single replacement, so the send must not be able to spend more than it granted.
677
+ async () => {
678
+ try {
679
+ return await send(activeRequest, "connection-reset", true);
680
+ } finally {
681
+ // Reselection can charge the rebuilt request after the initial send settled.
682
+ releaseRetainedRequest();
683
+ }
684
+ },
685
+ {
686
+ abortSignal: upstream.signal,
687
+ label: safeHostLabel(activeRequest.url),
688
+ attempts: remainingTransientSends(),
689
+ // The origin returned a head and may already be running the turn, so this is the
690
+ // ambiguous row of the stage table. Only the operator allowance the Responses stream
691
+ // also draws on can authorise it; without one the original failure stands.
692
+ authorize: () => authorizeResendForRecovery("headers-only", "connection-reset", ambiguousResend()).allowed,
693
+ // A replacement must satisfy the contract already promised to the client. A 200 that is
694
+ // not an event stream would reach the SSE translator as a non-SSE body and surface as a
695
+ // malformed-stream error instead of the reset that actually happened.
696
+ acceptResponse: replacement =>
697
+ (replacement.headers.get("content-type") ?? "").toLowerCase().includes("text/event-stream"),
698
+ },
699
+ );
700
+ const stream = nativeChatSse(resilientBody, {
633
701
  requestedModel,
634
702
  translatorBudget,
635
703
  signal: upstream.signal,
636
704
  stallTimeoutSec: config.stallTimeoutSec,
705
+ localUpstream: isLocalUpstream(activeProvider.baseUrl),
637
706
  onFirstOutput,
638
707
  onUsage: usage => {
639
708
  if (!recordKeyWireAttemptUsage(logCtx, usage)) {
@@ -18,6 +18,7 @@ import { sseFieldValue } from "../lib/sse-decoder";
18
18
  import { enforceAnthropicImageLimits, sniffImageDimensions } from "../adapters/anthropic-image-guard";
19
19
  import { normalizeAnthropicImages } from "../adapters/anthropic-image-normalize";
20
20
  import { createToolCallIdAllocator } from "../adapters/tool-call-id";
21
+ import { openAIChatSerializesThinking } from "../adapters/openai-chat/messages";
21
22
  import { messagesToResponsesTranslation } from "../protocols/codecs/messages";
22
23
  import { AnthropicRequestError, DesktopModelMappingUnavailableError, extractOcxEffortDirective, extractOcxRouteDirective, resolveInboundModel, type ClaudeCacheKeySource } from "../claude/inbound";
23
24
  import { isKnownDesktop3pModelId, resolveDesktop3pAlias } from "../claude/desktop-3p";
@@ -27,6 +28,7 @@ import { stripOneMillionMarker } from "../claude/context-windows";
27
28
  import { captureClaudeInbound } from "../claude/inbound-debug";
28
29
  import { claudeCodeForIngress } from "../claude/intercept/model-bindings";
29
30
  import { analyzeClaudeCompatibility, isClaudeCompatibilityMode } from "../claude/compatibility";
31
+ import { carriesMessageThread, messageThreadUnsupportedResponse } from "../claude/message-threads";
30
32
  import {
31
33
  applyReplayRefusalClientHeaders,
32
34
  carryReplayRefusal,
@@ -45,7 +47,12 @@ import {
45
47
  } from "../claude/outbound";
46
48
  import { clearableDeadline, idleDeadline } from "../lib/abort";
47
49
  import { estimateTokens } from "../lib/token-estimate";
48
- import { captureRouteStaticPolicy, NoEligiblePolicyCandidateError, UnknownRoutingPolicyError, routeModel } from "../router";
50
+ import {
51
+ CLAUDE_NATIVE_THINKING,
52
+ projectClaudeRequest,
53
+ type ClaudeThinkingProjection,
54
+ } from "../lib/claude-request-projection";
55
+ import { captureRouteStaticPolicy, NoEligiblePolicyCandidateError, previewRouteModel, routedProviderConfig, UnknownRoutingPolicyError, routeModel, type RouteResult } from "../router";
49
56
  import { evidenceFromBody } from "../routing/request-evidence";
50
57
  import { resolveWireProtocolOverride } from "./adapter-resolve";
51
58
  import type { OcxConfig } from "../types";
@@ -837,6 +844,12 @@ async function handleClaudeMessagesWithBudget(
837
844
  recordProtocolShadowPlan(logCtx, config, { inbound: "messages", model: requestedModel });
838
845
  return await anthropicNativePassthrough(req, config, logCtx, logIds, anthropicBody, "/v1/messages");
839
846
  }
847
+ // Only Anthropic holds message-thread state; the error makes Claude Code resend the full turn.
848
+ if (carriesMessageThread(anthropicBody)) {
849
+ logCtx.errorCode = "claude_thread_unsupported";
850
+ if (logIds) addFinalRequestLog(logIds.requestId, logIds.start, logCtx, 400, { closeReason: "non_stream" });
851
+ return messageThreadUnsupportedResponse();
852
+ }
840
853
  // Capture source semantics before effort rewriting or translation drops fields.
841
854
  // This policy is uniform across translated targets, including later fallback attempts.
842
855
  const compatibilityMode: unknown = config.claudeCode?.compatibility;
@@ -904,7 +917,7 @@ async function handleClaudeMessagesWithBudget(
904
917
  if (!requestedModel) requestedModel = (anthropicBody as Rec).model as string;
905
918
  const stream = internalBody.stream === true;
906
919
  /**
907
- * This proxy's count of the prompt it is about to forward, computed at most once.
920
+ * This proxy's count of the prompt it is about to forward, computed at most once per wire.
908
921
  *
909
922
  * Two readers want it and they want it under different rules. The usage log takes it as a
910
923
  * floor only for estimated-usage adapters, because its merge is `max(reported, estimate)` and
@@ -913,12 +926,17 @@ async function handleClaudeMessagesWithBudget(
913
926
  * `message_delta` still corrects it (#4857).
914
927
  */
915
928
  let requestTokenFloor: number | undefined;
916
- const claudeRequestTokenFloor = (): number => {
917
- if (requestTokenFloor === undefined) {
918
- requestTokenFloor = estimateClaudeRequestTokens(anthropicBody as Rec, requestedModel);
919
- }
920
- return requestTokenFloor;
921
- };
929
+ /**
930
+ * The `text|signature|redacted` triple `requestTokenFloor` was measured under.
931
+ *
932
+ * The count is not a constant for the request: a combo re-picks its child at dispatch and a
933
+ * retry can rotate the wire, so the pre-dispatch callers and the post-dispatch translator can
934
+ * legitimately want different projections. Keying the memo re-measures when they disagree
935
+ * instead of handing the translator the pre-dispatch measurement, and the estimator still runs
936
+ * at most twice for one request. At most, because a request has one settled wire per phase and
937
+ * the key collapses every repeat of the same answer.
938
+ */
939
+ let requestTokenFloorKey: string | undefined;
922
940
  // Routed adapters only support streamed turns; always stream internally and fold
923
941
  // the translated Anthropic SSE into a message JSON for non-streaming clients.
924
942
  internalBody.stream = true;
@@ -927,7 +945,25 @@ async function handleClaudeMessagesWithBudget(
927
945
  // Native ChatGPT passthrough (openai-responses forward) accepts only Codex-shaped
928
946
  // bodies: it 400s on sampling params ("Unsupported parameter: max_output_tokens",
929
947
  // verified live 2026-07-11). Strip them for that route; routed providers keep them.
930
- let settledRoute: ReturnType<typeof routeModel> | undefined;
948
+ let settledRoute: RouteResult | undefined;
949
+ /**
950
+ * Which replayed thinking fields the send that actually happened will serialize.
951
+ *
952
+ * Read lazily: routing settles the ingress wire, core may then re-pick it (a combo child, a
953
+ * rotated retry), and this is called both before and after that happens. It therefore reads the
954
+ * physical attempt when one exists and the ingress route only until then.
955
+ */
956
+ const claudeThinkingProjection = (): ClaudeThinkingProjection =>
957
+ thinkingProjectionForDispatch(config, settledRoute, logCtx);
958
+ const claudeRequestTokenFloor = (): number => {
959
+ const thinking = claudeThinkingProjection();
960
+ const key = `${thinking.text}|${thinking.signature}|${thinking.redacted}`;
961
+ if (requestTokenFloor === undefined || requestTokenFloorKey !== key) {
962
+ requestTokenFloor = estimateClaudeRequestTokens(anthropicBody as Rec, requestedModel, thinking);
963
+ requestTokenFloorKey = key;
964
+ }
965
+ return requestTokenFloor;
966
+ };
931
967
  try {
932
968
  const route = routeModel(config, internalBody.model as string, evidenceFromBody(internalBody));
933
969
  // Same reason as the native Chat lane: this route can be sent from here, so
@@ -1330,12 +1366,20 @@ function estimateBase64AttachmentTokens(data: string): number {
1330
1366
  * characters: one 2MB screenshot is ~2.7M base64 chars, which the plain chars/token
1331
1367
  * divide reports as hundreds of thousands of tokens versus a real cost around 1.6k.
1332
1368
  * That breaks the >2x drift bound the estimator is held to (devlog 260711_claude_inbound
1333
- * 040 §3). Text and url sources are left in place and counted as characters, as is
1369
+ * 040 §3); a live 260-message turn whose replayed thinking was 78.8% of the body breached
1370
+ * it at 3.28x, which is why the estimate is projected onto the settled route. Text and url
1371
+ * sources are left in place and counted as characters, as is
1334
1372
  * anything outside protocol content positions (tool_use.input, tool schemas).
1373
+ *
1374
+ * `thinking` selects which replayed thinking fields the SETTLED route serializes, so the measure
1375
+ * describes the prompt this proxy forwards rather than the one the caller typed. Omitted, the
1376
+ * whole body counts — correct for the Anthropic-native wire, where nothing is projected away.
1377
+ * See `claude-request-projection.ts` for why a routed wire must project it out.
1335
1378
  */
1336
1379
  export function estimateClaudeRequestTokens(
1337
1380
  raw: { system?: unknown; messages?: unknown; tools?: unknown },
1338
1381
  modelId: string | undefined,
1382
+ thinking: ClaudeThinkingProjection = CLAUDE_NATIVE_THINKING,
1339
1383
  ): number {
1340
1384
  let attachmentTokens = 0;
1341
1385
  // Blank base64 payloads ONLY in protocol content positions: message content blocks and
@@ -1370,11 +1414,93 @@ export function estimateClaudeRequestTokens(
1370
1414
  : messages;
1371
1415
  const parts: string[] = [];
1372
1416
  if (raw.system !== undefined) parts.push(typeof raw.system === "string" ? raw.system : JSON.stringify(raw.system));
1373
- if (raw.messages !== undefined) parts.push(JSON.stringify(sanitizedMessages(raw.messages)));
1417
+ if (raw.messages !== undefined) {
1418
+ const projected = projectClaudeRequest(raw, thinking);
1419
+ parts.push(JSON.stringify(sanitizedMessages(projected.messages)));
1420
+ }
1374
1421
  if (raw.tools !== undefined) parts.push(JSON.stringify(raw.tools));
1375
1422
  return Math.max(1, estimateTokens(parts.join("\n"), modelId) + attachmentTokens);
1376
1423
  }
1377
1424
 
1425
+ /**
1426
+ * The projection for a route that has already settled.
1427
+ *
1428
+ * Only the OpenAI-shaped Chat adapter discards replayed thinking; every other settled wire
1429
+ * forwards the body it was given. An unknown route keeps the full body.
1430
+ */
1431
+ function thinkingProjectionForRoute(route: RouteResult | undefined): ClaudeThinkingProjection {
1432
+ if (!route || route.provider.adapter !== "openai-chat") return CLAUDE_NATIVE_THINKING;
1433
+ return openAIChatSerializesThinking(route.provider, route.modelId);
1434
+ }
1435
+
1436
+ /**
1437
+ * The projection for the wire that will physically carry the request.
1438
+ *
1439
+ * The ingress route is not the last word on that wire. A combo re-picks its child at dispatch,
1440
+ * and a retry can rotate the adapter mid-turn, so the ingress pick can name a different provider
1441
+ * — and therefore a different body — than the one that is sent. `logCtx.activeAttempt` is the
1442
+ * send that actually happened (`sealRequestAttemptIdentity` keeps its adapter current), so it
1443
+ * wins once it exists; before the first send the ingress route is the only authority there is.
1444
+ *
1445
+ * A Chat identity is re-derived through `routedProviderConfig`, not read off the raw config row:
1446
+ * `preserveReasoningContentModels` is registry-merged, so a row that omits it would otherwise
1447
+ * price a preserve-listed model as if the wire dropped its reasoning.
1448
+ */
1449
+ function thinkingProjectionForDispatch(
1450
+ config: OcxConfig,
1451
+ route: RouteResult | undefined,
1452
+ logCtx: Pick<RequestLogContext, "providerAdapter" | "activeAttempt">,
1453
+ ): ClaudeThinkingProjection {
1454
+ const attempt = logCtx.activeAttempt;
1455
+ const adapter = attempt?.adapter ?? logCtx.providerAdapter;
1456
+ // No send to describe yet: the ingress route is the best available answer, and for the
1457
+ // Anthropic-native wire (which drops nothing) it is already the right one.
1458
+ if (adapter === undefined) return thinkingProjectionForRoute(route);
1459
+ if (adapter !== "openai-chat") return CLAUDE_NATIVE_THINKING;
1460
+ const providerName = attempt?.provider;
1461
+ const modelId = attempt?.model ?? route?.modelId;
1462
+ const provider = providerName !== undefined && Object.hasOwn(config.providers, providerName)
1463
+ ? config.providers[providerName]
1464
+ : undefined;
1465
+ // A Chat wire whose destination cannot be named keeps the route's own answer: over-counting on
1466
+ // a path that publishes nothing is harmless, under-counting a real prompt is not.
1467
+ if (providerName === undefined || modelId === undefined || provider === undefined) {
1468
+ return thinkingProjectionForRoute(route);
1469
+ }
1470
+ try {
1471
+ return openAIChatSerializesThinking(routedProviderConfig(providerName, provider), modelId);
1472
+ } catch {
1473
+ return thinkingProjectionForRoute(route);
1474
+ }
1475
+ }
1476
+
1477
+ /**
1478
+ * The projection for a route resolved only to MEASURE a body this handler never sends.
1479
+ *
1480
+ * `previewRouteModel` is the read-only resolver: it advances no combo round-robin state, so a
1481
+ * count request cannot steer where the next real turn goes. An unresolvable model keeps the
1482
+ * full body, matching the old behavior for models routing cannot place.
1483
+ *
1484
+ * The wire is settled exactly as the turn path settles it. Routing fills in the provider's
1485
+ * registry adapter, and a per-model `modelAdapters` override or a pinned wire is applied later by
1486
+ * `resolveWireProtocolOverride` — so skipping it here would price the count against a body the
1487
+ * routed adapter never sends.
1488
+ */
1489
+ function thinkingProjectionForPreview(config: OcxConfig, modelId: string): ClaudeThinkingProjection {
1490
+ try {
1491
+ const route = previewRouteModel(config, modelId);
1492
+ route.staticPolicy = captureRouteStaticPolicy(
1493
+ route.providerName, route.modelId, route.provider, route.staticPolicy.effectiveAlias, "anthropic",
1494
+ );
1495
+ route.provider = resolveWireProtocolOverride(
1496
+ route.providerName, route.modelId, route.provider, "anthropic", route.staticPolicy,
1497
+ );
1498
+ return thinkingProjectionForRoute(route);
1499
+ } catch {
1500
+ return CLAUDE_NATIVE_THINKING;
1501
+ }
1502
+ }
1503
+
1378
1504
  export async function handleClaudeCountTokens(
1379
1505
  req: Request,
1380
1506
  config: OcxConfig,
@@ -1431,11 +1557,20 @@ export async function handleClaudeCountTokens(
1431
1557
  if (wantsNativePassthrough(req, config, requestPolicy, model, cc)) {
1432
1558
  return await anthropicNativePassthrough(req, config, { model, provider: "anthropic-native", surface: "claude" }, undefined, raw, "/v1/messages/count_tokens");
1433
1559
  }
1560
+ // A thread delta would undercount; refuse it exactly as the translated Messages path does.
1561
+ if (carriesMessageThread(raw)) return messageThreadUnsupportedResponse();
1434
1562
  // PF-08: an eligible managed-key route counts the body the native lane would send.
1435
1563
  const nativeCountBody = resolveProtocolSettings(config).rollout.managedMessagesNative
1436
1564
  ? (await import("./messages-native")).nativeMessagesCountBody(config, cc, raw, { fastRow: countFastRow !== null })
1437
1565
  : undefined;
1438
- const inputTokens = estimateClaudeRequestTokens(nativeCountBody ?? raw, model);
1566
+ // A count answers for the prompt a real turn from this model would forward, so it projects
1567
+ // the same unserialized content that turn's `message_start` floor does. Counting the raw
1568
+ // caller body instead reported replayed thinking this route never sends (#4857 family).
1569
+ const inputTokens = estimateClaudeRequestTokens(
1570
+ nativeCountBody ?? raw,
1571
+ model,
1572
+ thinkingProjectionForPreview(config, model),
1573
+ );
1439
1574
  return new Response(JSON.stringify({ input_tokens: inputTokens }), {
1440
1575
  status: 200,
1441
1576
  headers: { "Content-Type": "application/json" },
@@ -29,6 +29,10 @@ export type LinkListenerStatus = {
29
29
  reason: string | null;
30
30
  };
31
31
 
32
+ export function linkListenerOwnsTarget(status: LinkListenerStatus): boolean {
33
+ return status.state === "listening" && status.port !== null;
34
+ }
35
+
32
36
  export interface LinkListenerLifecycle<T> {
33
37
  ownsListener(server: Server<T>): boolean;
34
38
  start(ctx: LinkListenerStartContext<T>): void;
@@ -93,6 +97,9 @@ export function createLinkListenerLifecycle<T>(deps: LinkListenerDeps = {}): Lin
93
97
  bound = serve({
94
98
  hostname: "127.0.0.1",
95
99
  port: requestedPort,
100
+ // The public listener's idle limit (serve-options.ts). Bun's 10 s default would cut a
101
+ // relayed turn that the Home holds or that streams with a long gap.
102
+ idleTimeout: 255,
96
103
  maxRequestBodySize: startContext.maxRequestBodySize,
97
104
  fetch: (req: Request, server: Server<unknown>) => startContext!.dispatch(req, server as Server<T>),
98
105
  } as Parameters<typeof Bun.serve>[0]);
@@ -6,6 +6,7 @@ import {
6
6
  } from "./claude-intercept-lifecycle";
7
7
  import {
8
8
  createLinkListenerLifecycle,
9
+ linkListenerOwnsTarget,
9
10
  linkRouteAllowed,
10
11
  type LinkListenerDeps,
11
12
  type LinkListenerLifecycle,
@@ -82,7 +83,9 @@ export function createOptionalListenerSet<T>(linkDeps: LinkListenerDeps = {}): O
82
83
  activeConfig = ctx.config;
83
84
  linkListener.start({ dispatch: ctx.dispatch, maxRequestBodySize: ctx.maxRequestBodySize });
84
85
  unregisterSupervisorAdmission ??= linkListener.onAuthenticatedCatalog(apiKeyId => supervisor.notifyAuthenticatedRequest?.(apiKeyId));
85
- supervisor.start();
86
+ if (linkListenerOwnsTarget(linkListener.status())) {
87
+ supervisor.start();
88
+ }
86
89
  supervisorStop = () => supervisor.stop();
87
90
  claudeIntercept.start({
88
91
  config: ctx.config,
@@ -2,9 +2,11 @@ import {
2
2
  createRuntimePackageTreeIntegrityGuard,
3
3
  type PackageTreeIntegrityGuard,
4
4
  } from "../../lib/package-tree-integrity";
5
+ import { createPackageTreeRetargetWatch } from "../../lib/package-tree-retarget";
5
6
  import { acceptSystemRestart } from "../management/system-restart";
6
7
  import { inspectNativeCodexOwnership } from "../../integrations/native/ownership-preflight";
7
8
  import { detectInstall } from "../../update/index";
9
+ import { planMiseLauncherTargetWatch } from "../../update/mise-launcher-target";
8
10
  import type { StartServerDeps } from "./startup-warnings";
9
11
 
10
12
  // Production guard wiring for the package-tree integrity fence. Tests inject
@@ -21,31 +23,48 @@ export function createPackageTreeIntegrityGuardForServer(
21
23
  }
22
24
  const acceptPackageTreeRestart = deps.acceptSystemRestart ?? acceptSystemRestart;
23
25
  let vetoAcceptedRestart: (() => void) | undefined;
26
+ // An out-of-band install replaced or retargeted the package this live process runs. Let the
27
+ // standard drain-and-restart path bring the new tree up instead of refusing traffic until a
28
+ // manual restart. acceptSystemRestart is idempotent and supervisor-aware. Throwing tells the
29
+ // caller to retry after its own debounce.
30
+ const restartOntoNewPackageTree = () => {
31
+ const beforeScheduledDrain = () => !isServiceChild() || serviceHomeOwned();
32
+ if (!beforeScheduledDrain()) throw new Error("service home ownership changed");
33
+ let admitted = false;
34
+ const result = acceptPackageTreeRestart(undefined, {
35
+ onAccepted: veto => { admitted = true; vetoAcceptedRestart = veto; },
36
+ beforeScheduledDrain,
37
+ });
38
+ // An automatic restart that could not take the drain lease (another drain owns the lifecycle
39
+ // gate) reports alreadyDraining without admitting it. Throw so the caller retries instead of
40
+ // treating the restart as done and leaving the service on the old package.
41
+ if (result.alreadyDraining && !admitted) throw new Error("restart admission is busy");
42
+ };
24
43
  const guard = createRuntimePackageTreeIntegrityGuard(
25
44
  deps.packageTreeInstaller ?? detectInstall(),
26
45
  deps.observePackageTree,
27
46
  undefined,
28
- {
29
- ...deps.packageTreeIntegrityOptions,
30
- onReplaced: () => {
31
- // An out-of-band install replaced the package under this live process. Serve
32
- // the 503 for the triggering request, then let the standard drain-and-restart
33
- // path bring the new tree up instead of refusing traffic until a manual
34
- // restart. acceptSystemRestart is idempotent and supervisor-aware.
35
- const beforeScheduledDrain = () => !isServiceChild() || serviceHomeOwned();
36
- if (!beforeScheduledDrain()) throw new Error("service home ownership changed");
37
- acceptPackageTreeRestart(undefined, {
38
- onAccepted: veto => { vetoAcceptedRestart = veto; },
39
- beforeScheduledDrain,
40
- });
41
- },
42
- },
47
+ { ...deps.packageTreeIntegrityOptions, onReplaced: restartOntoNewPackageTree },
43
48
  );
49
+ // mise installs each version beside the last and repoints a floating link, so the manifest
50
+ // above never changes on `mise upgrade`. The managed service follows its launcher instead.
51
+ const plan = deps.packageTreeLauncherTarget === undefined
52
+ ? planMiseLauncherTargetWatch()
53
+ : deps.packageTreeLauncherTarget;
54
+ const retarget = plan
55
+ ? createPackageTreeRetargetWatch(
56
+ plan.runningRoot,
57
+ plan.resolveTarget,
58
+ restartOntoNewPackageTree,
59
+ deps.packageTreeRetargetOptions,
60
+ )
61
+ : null;
44
62
  return {
45
63
  status: () => guard.status(),
46
- installedVersion: () => guard.installedVersion?.(),
64
+ installedVersion: () => guard.installedVersion?.() ?? retarget?.settledVersion(),
47
65
  dispose: () => {
48
66
  guard.dispose();
67
+ retarget?.dispose();
49
68
  vetoAcceptedRestart?.();
50
69
  vetoAcceptedRestart = undefined;
51
70
  },
@@ -87,6 +87,7 @@ import {
87
87
  import { sessionLaneIdFromRequest } from "../request-log-conversation";
88
88
  import { responseWithDeferredRequestLog } from "../relay";
89
89
  import { createRequestMetricsOwner } from "../request-metrics";
90
+ import { cachedKiroQuotaMetricRows } from "../../providers/kiro-quota-metrics";
90
91
  import {
91
92
  corsHeaders,
92
93
  managementCorsHeaders,
@@ -292,7 +293,8 @@ export function createServeOptions(ctx: ServeOptionsContext) {
292
293
  port,
293
294
  } = ctx;
294
295
  void port;
295
- const requestMetrics = metricsExportEnabled(config) ? createRequestMetricsOwner() : undefined;
296
+ const requestMetrics = metricsExportEnabled(config)
297
+ ? createRequestMetricsOwner(Date.now() / 1000, cachedKiroQuotaMetricRows) : undefined;
296
298
  const requestMetricsLogContext = requestMetrics ? { requestMetricsRecorder: requestMetrics } : {};
297
299
  const requestManagementApiDeps: ManagementApiDeps = requestMetrics
298
300
  ? { ...managementApiDeps, requestMetrics: { snapshot: () => requestMetrics.snapshot() } }
@@ -1398,7 +1400,7 @@ export function createServeOptions(ctx: ServeOptionsContext) {
1398
1400
  };
1399
1401
  return runAdmittedHttpTurn(req, policy, async turnAdmissionLease => {
1400
1402
  const response = await handleContextHistory(req, config, logCtx, contextEndpoint(url.pathname)!,
1401
- turnAdmissionLease, admission, () => resolveApiAuth(req, policy));
1403
+ turnAdmissionLease, admission, () => resolveApiAuth(req, ingress === "hub-link" ? linkPolicy() : policy));
1402
1404
  addFinalRequestLog(requestId, start, logCtx, response.status,
1403
1405
  response.status === 499 ? { closeReason: "client_cancel" } : undefined);
1404
1406
  return withCors(response, req, policy);
@@ -11,6 +11,8 @@ import type {
11
11
  PackageTreeIntegrityOptions,
12
12
  PackageTreeRuntimeInstall,
13
13
  } from "../../lib/package-tree-integrity";
14
+ import type { PackageTreeRetargetOptions } from "../../lib/package-tree-retarget";
15
+ import type { MiseLauncherTargetWatchPlan } from "../../update/mise-launcher-target";
14
16
  import {
15
17
  consumeForInspection,
16
18
  relaySseWithHeartbeat,
@@ -158,6 +160,10 @@ export interface StartServerDeps {
158
160
  packageTreeServiceChild?: () => boolean;
159
161
  /** Test-only: whether this service child still owns its service home. */
160
162
  packageTreeServiceHomeOwned?: () => boolean;
163
+ /** Test-only launcher plan; production plans from the mise owner and service state. Null disables. */
164
+ packageTreeLauncherTarget?: MiseLauncherTargetWatchPlan | null;
165
+ /** Test-only retarget-watch timing and version seams. */
166
+ packageTreeRetargetOptions?: PackageTreeRetargetOptions;
161
167
  /** Test-only seam for observing quota-worker registration ownership. */
162
168
  registerCodexQuotaAutoRefreshWorker?: typeof registerCodexQuotaAutoRefreshWorker;
163
169
  }