@bitkyc08/opencodex 2.49.0 → 2.51.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. package/AGENTS_INSTALL.md +9 -1
  2. package/README.md +3 -0
  3. package/bin/ocx.mjs +222 -71
  4. package/gui/dist/assets/index-D7BdZpZm.js +115 -0
  5. package/gui/dist/index.html +1 -1
  6. package/package.json +1 -1
  7. package/src/adapters/qoder/adapter.ts +69 -1
  8. package/src/adapters/qoder/scaffold-guard.ts +233 -0
  9. package/src/claude/agents-inject.ts +29 -5
  10. package/src/claude/desktop-3p.ts +31 -3
  11. package/src/claude/gateway-cache.ts +12 -21
  12. package/src/claude/inbound.ts +17 -5
  13. package/src/cli/account-api.ts +18 -3
  14. package/src/cli/account-auth.ts +8 -1
  15. package/src/cli/account-extended.ts +2 -1
  16. package/src/cli/account.ts +1 -0
  17. package/src/cli/capabilities.ts +43 -1
  18. package/src/cli/claude-agent-startup-sync.ts +26 -1
  19. package/src/cli/claude.ts +138 -20
  20. package/src/cli/config-command.ts +67 -1
  21. package/src/cli/connect.ts +181 -14
  22. package/src/cli/dispatch.ts +53 -9
  23. package/src/cli/doctor.ts +9 -2
  24. package/src/cli/ensure-desired-integrations.ts +10 -0
  25. package/src/cli/gui-pair-client.ts +1 -12
  26. package/src/cli/help.ts +4 -1
  27. package/src/cli/hub.ts +367 -0
  28. package/src/cli/index.ts +99 -31
  29. package/src/cli/launcher-context.ts +1 -1
  30. package/src/cli/models-runtime.ts +8 -3
  31. package/src/cli/observe.ts +13 -3
  32. package/src/cli/registry.ts +43 -3
  33. package/src/cli/status.ts +325 -5
  34. package/src/cli/version-skew.ts +4 -1
  35. package/src/cli.ts +2 -2
  36. package/src/client/catalog-compatibility.ts +192 -0
  37. package/src/client/connect.ts +31 -0
  38. package/src/client/hub-client.ts +52 -0
  39. package/src/client/hub-state.ts +214 -0
  40. package/src/clients/config-export/zcode.ts +24 -0
  41. package/src/codex/account-runtime-state.ts +6 -1
  42. package/src/codex/account-store.ts +72 -9
  43. package/src/codex/account-usability.ts +50 -13
  44. package/src/codex/auth-api.ts +156 -28
  45. package/src/codex/auth-context.ts +21 -0
  46. package/src/codex/catalog/effort.ts +67 -8
  47. package/src/codex/catalog/parsing.ts +23 -0
  48. package/src/codex/catalog/provider-fetch.ts +71 -2
  49. package/src/codex/catalog/sync.ts +99 -0
  50. package/src/codex/codex-write-lock.ts +11 -2
  51. package/src/codex/desired-state.ts +47 -1
  52. package/src/codex/inject-coordination.ts +10 -5
  53. package/src/codex/inject.ts +29 -12
  54. package/src/codex/loopback-target.ts +45 -0
  55. package/src/codex/quota-auto-refresh.ts +6 -1
  56. package/src/codex/quota.ts +54 -8
  57. package/src/codex/routing.ts +48 -1
  58. package/src/codex/runtime.ts +37 -3
  59. package/src/codex/sync.ts +29 -9
  60. package/src/codex/warmup.ts +21 -4
  61. package/src/combos/index.ts +2 -0
  62. package/src/combos/resolve.ts +52 -0
  63. package/src/config/pending-teardown.ts +1 -1
  64. package/src/config.ts +184 -12
  65. package/src/generated/compatibility-version.json +188 -116
  66. package/src/grok/status.ts +9 -1
  67. package/src/integrations/config-io.ts +54 -1
  68. package/src/lib/bun-runtime.ts +1 -1
  69. package/src/lib/errors.ts +8 -0
  70. package/src/lib/gui-pair-capability.ts +27 -0
  71. package/src/lib/local-destinations.ts +162 -0
  72. package/src/lib/package-tree-integrity.ts +1 -1
  73. package/src/lib/privacy.ts +25 -0
  74. package/src/lib/process-control.ts +130 -20
  75. package/src/lib/service-secrets.ts +28 -0
  76. package/src/lib/test-home-guard.ts +49 -0
  77. package/src/oauth/health.ts +47 -12
  78. package/src/oauth/index.ts +46 -8
  79. package/src/oauth/token-guardian.ts +32 -6
  80. package/src/providers/google-ai-studio-model-discovery.ts +74 -0
  81. package/src/providers/opencode-go-transport.ts +9 -1
  82. package/src/providers/opencode-zen-rate-limit.ts +75 -0
  83. package/src/providers/quota.ts +20 -1
  84. package/src/providers/registry.ts +35 -6
  85. package/src/remote/hub-state.ts +182 -0
  86. package/src/server/auth-cors.ts +11 -0
  87. package/src/server/chat-completions.ts +10 -7
  88. package/src/server/chat-native.ts +10 -1
  89. package/src/server/claude-messages.ts +12 -6
  90. package/src/server/hub-state.ts +98 -0
  91. package/src/server/images.ts +2 -2
  92. package/src/server/index.ts +149 -8
  93. package/src/server/management/api-access.ts +14 -3
  94. package/src/server/management/config-routes.ts +2 -2
  95. package/src/server/management/cursor-integration-routes.ts +13 -4
  96. package/src/server/management/logs-usage-routes.ts +4 -1
  97. package/src/server/management/model-rows.ts +16 -1
  98. package/src/server/management/oauth-account-routes.ts +6 -2
  99. package/src/server/management/provider-routes.ts +9 -2
  100. package/src/server/management/request-history-routes.ts +4 -2
  101. package/src/server/management/route-registry.ts +5 -4
  102. package/src/server/management/shared.ts +66 -3
  103. package/src/server/management-api.ts +1 -1
  104. package/src/server/proxy-liveness.ts +7 -1
  105. package/src/server/request-decompress.ts +91 -3
  106. package/src/server/request-log-conversation.ts +41 -1
  107. package/src/server/request-log.ts +10 -0
  108. package/src/server/responses/codex-auth-error.ts +18 -1
  109. package/src/server/responses/codex-ws-exchange.ts +36 -4
  110. package/src/server/responses/codex-ws-wire.ts +76 -5
  111. package/src/server/responses/compact.ts +28 -11
  112. package/src/server/responses/context-overflow.ts +11 -0
  113. package/src/server/responses/core.ts +201 -48
  114. package/src/server/responses/policy-fallback.ts +13 -3
  115. package/src/server/search.ts +2 -2
  116. package/src/server/system-env-shell.ts +14 -2
  117. package/src/server/system-env.ts +106 -14
  118. package/src/service.ts +965 -68
  119. package/src/types/accounts.ts +18 -0
  120. package/src/types/config.ts +93 -4
  121. package/src/types/provider.ts +56 -0
  122. package/src/types.ts +4 -0
  123. package/src/update/badge.ts +3 -2
  124. package/src/update/index.ts +317 -64
  125. package/src/update/install-detection.d.mts +6 -0
  126. package/src/update/install-detection.mjs +73 -0
  127. package/src/update/job.ts +101 -49
  128. package/src/update/pnpm-global-install.d.mts +144 -0
  129. package/src/update/pnpm-global-install.mjs +591 -0
  130. package/src/update/pnpm-invocation.d.mts +43 -0
  131. package/src/update/pnpm-invocation.mjs +141 -0
  132. package/src/update/registry-integrity.d.mts +16 -0
  133. package/src/update/registry-integrity.mjs +37 -0
  134. package/src/update/transactional-install.d.mts +1 -1
  135. package/src/update/transactional-install.mjs +101 -7
  136. package/src/update/tray-update-plan.mjs +1 -1
  137. package/src/vision/plan.ts +13 -3
  138. package/src/vision/routed-describe.ts +51 -20
  139. package/src/web-search/ollama-executor.ts +127 -0
  140. package/src/web-search/passthrough-bridge.ts +761 -0
  141. package/gui/dist/assets/index-BtyONQrZ.js +0 -115
@@ -17,6 +17,7 @@ import {
17
17
  loadConfig,
18
18
  saveConfig,
19
19
  getConfigDir,
20
+ loopbackCompanionBindError,
20
21
  websocketsEnabled,
21
22
  } from "../config";
22
23
  import { grokDefaultReasoningEffort } from "../grok/effort";
@@ -28,6 +29,7 @@ import { withCatalogWriteSerialization } from "../codex/catalog-write-serializat
28
29
  import { invalidateCodexModelsCacheWithPermit } from "../codex/catalog/sync";
29
30
  import { currentServiceHomes, serviceStatePathsForOpenCodexHome } from "../service";
30
31
  import { shouldSyncCodexOnStart } from "../codex/desired-state";
32
+ import { effectiveLoopbackListenerPort } from "../codex/loopback-target";
31
33
  import {
32
34
  createWindowsTaskListingCache,
33
35
  inspectNativeCodexOwnership,
@@ -63,7 +65,11 @@ import { runModelRenameStartupMigration } from "../providers/model-rename-startu
63
65
  import { isCanonicalOpenAiForwardProvider, OPENAI_CODEX_PROVIDER_ID } from "../providers/openai-tiers";
64
66
  import { providerCodexAccountMode } from "../providers/registry";
65
67
  import type { StorageCleanupPolicy } from "../types";
66
- import { MAX_DECOMPRESSED_BODY_BYTES } from "./request-decompress";
68
+ import {
69
+ MAX_CONFIGURABLE_INBOUND_BODY_BYTES,
70
+ MIN_CONFIGURABLE_INBOUND_BODY_BYTES,
71
+ resolveInboundBodyLimitBytes,
72
+ } from "./request-decompress";
67
73
  import {
68
74
  CodexAccountCooldownError,
69
75
  cooldownErrorMessage,
@@ -769,8 +775,16 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
769
775
  const bindHost = !configuredHost || /^localhost$/i.test(configuredHost) ? "127.0.0.1" : configuredHost;
770
776
 
771
777
  // Unauthenticated loopback listener (#1102). Off unless explicitly enabled.
778
+ // A port-less enabled entry is the companion form: same port as the public listener, on
779
+ // 127.0.0.1 (#4236). Refuse an impossible pair here, before any bind, so a hand edit that
780
+ // bypassed validateConfigCandidate reports the collision rather than EADDRINUSE from a
781
+ // rollback that looks like a foreign process holding the port.
772
782
  const loopbackListener = config.unauthenticatedLoopbackListener;
773
- const loopbackListenerPort = loopbackListener?.enabled ? loopbackListener.port : null;
783
+ if (loopbackListener?.enabled === true && loopbackListener.port === undefined) {
784
+ const companionError = loopbackCompanionBindError(config.hostname, listenPort);
785
+ if (companionError) throw new Error(companionError);
786
+ }
787
+ const loopbackListenerPort = effectiveLoopbackListenerPort(config, listenPort);
774
788
  // Hub management ingress is a third, management-only listener. Its address is intentionally
775
789
  // fixed: the kernel loopback bind is the trust boundary that permits Tailscale identity headers.
776
790
  const managementIngress = config.runtimeRole === "hub" ? config.hub?.managementIngress : undefined;
@@ -808,6 +822,25 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
808
822
  * keeps the paid upstream behind its own admission and forward-credential checks, so admit only
809
823
  * the exact methods and paths it serves (#3428).
810
824
  *
825
+ * `POST /v1/messages` (Anthropic wire) and `POST /v1/chat/completions` (OpenAI chat wire)
826
+ * are the inference endpoints the hub's OWN local clients speak: `ocx claude` and the
827
+ * `system-env` injection and Claude Desktop 3P dial the first, Cursor Private Inference, the
828
+ * vision `routed-describe` helper and aside/opencode the second (#4236). On a hub whose
829
+ * public listener binds a tailnet address there is no other local socket for them, so
830
+ * leaving them off this list left every non-Codex local client pointed at a closed port.
831
+ * Both handlers resolve their own admission from the RECEIVING listener's policy view — the
832
+ * same resolver and the same loopback short-circuit `/v1/responses` already uses — so this
833
+ * adds a wire, not a trust level. `/api/*` is deliberately still absent: local management
834
+ * discovery goes to the authenticated management surface, never to this listener.
835
+ *
836
+ * `POST /v1/messages/count_tokens` completes that Anthropic wire. It is admitted on a
837
+ * narrower argument than the other two rather than on symmetry: it spends no provider quota,
838
+ * reaches no stored credential, and returns a token count computed from the request body the
839
+ * caller already holds. Withholding it bought no confinement — the same caller may POST the
840
+ * whole conversation to `/v1/messages` on this socket — and cost Claude Code its server-side
841
+ * count, which it then silently replaces with a local estimate. `/api/*`, `/healthz`,
842
+ * `/readyz` and the GUI remain 404 here, which is the boundary that actually matters.
843
+ *
811
844
  * `GET /v1/models` is on the list for a reason that is easy to miss. When catalog
812
845
  * materialization fails or finds no source, `syncCodex` warns and injects with
813
846
  * `catalogPath: null`; Codex then builds an ONLINE model manager and `model/list` refreshes
@@ -820,6 +853,8 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
820
853
  return req.method === "POST" || req.headers.get("upgrade")?.toLowerCase() === "websocket";
821
854
  }
822
855
  if (path === "/v1/responses/compact") return req.method === "POST";
856
+ if (path === "/v1/messages" || path === "/v1/chat/completions") return req.method === "POST";
857
+ if (path === "/v1/messages/count_tokens") return req.method === "POST";
823
858
  if (path === "/v1/alpha/search") return req.method === "POST";
824
859
  if (path === "/v1/images/generations" || path === "/v1/images/edits") {
825
860
  return req.method === "POST";
@@ -1023,6 +1058,22 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
1023
1058
  let loopbackServer: Server<WsData> | null = null;
1024
1059
  let managementIngressServer: Server<WsData> | null = null;
1025
1060
 
1061
+ // Resolved once, before any listener binds. The clamp is silent inside the resolver so it
1062
+ // stays pure and per-request cheap; the operator is told here instead, once, because a
1063
+ // config value that was quietly reduced is exactly the thing they would otherwise debug
1064
+ // against the wrong limit.
1065
+ const inboundBodyLimitBytes = resolveInboundBodyLimitBytes(config.maxInboundBodyBytes);
1066
+ const requestedInboundBodyLimit = config.maxInboundBodyBytes;
1067
+ if (requestedInboundBodyLimit !== undefined
1068
+ && requestedInboundBodyLimit > 0
1069
+ && requestedInboundBodyLimit !== inboundBodyLimitBytes) {
1070
+ console.warn(
1071
+ `[server] maxInboundBodyBytes=${requestedInboundBodyLimit} is outside the supported range `
1072
+ + `[${MIN_CONFIGURABLE_INBOUND_BODY_BYTES}, ${MAX_CONFIGURABLE_INBOUND_BODY_BYTES}]; `
1073
+ + `using ${inboundBodyLimitBytes} bytes.`,
1074
+ );
1075
+ }
1076
+
1026
1077
  type ServerIngress = "public" | "unauthenticated-loopback" | "hub-management";
1027
1078
  function ingressForServer(requestServer: Server<WsData>): ServerIngress {
1028
1079
  if (requestServer === loopbackServer) return "unauthenticated-loopback";
@@ -1042,7 +1093,10 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
1042
1093
  userCostOverlayReconciler = startUserCostOverlayReconciler({ liveConfig: config });
1043
1094
  const serveOptions = {
1044
1095
  idleTimeout: 255,
1045
- maxRequestBodySize: MAX_DECOMPRESSED_BODY_BYTES,
1096
+ // Bun rejects an oversized body before `fetch` runs, so the listener has to be raised
1097
+ // with the admission limit or the opt-in would do nothing. Fixed at bind time: a live
1098
+ // `maxInboundBodyBytes` edit needs a restart, which the config doc states.
1099
+ maxRequestBodySize: inboundBodyLimitBytes,
1046
1100
  async fetch(req: Request, requestServer: Server<WsData>): Promise<Response> {
1047
1101
  const ingress = ingressForServer(requestServer);
1048
1102
  // The unauthenticated loopback listener (#1102) serves a fixed allowlist and nothing
@@ -1347,6 +1401,83 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
1347
1401
  );
1348
1402
  }
1349
1403
 
1404
+ if (url.pathname === "/v1/hub-state" && (req.method === "GET" || req.method === "HEAD")) {
1405
+ // #4236: a connected client had no way to learn which providers this hub can actually
1406
+ // serve, so `ocx status` on the client reported the CLIENT's empty credential store as
1407
+ // if it were the truth — "xai ✗ not logged in" on a machine whose hub has xAI logged
1408
+ // in. The fix is one least-privilege data-plane read, in the /v1/catalog (#809)
1409
+ // tradition: same admission resolver, same origin check, no parameters, no caller
1410
+ // credential forwarded upstream, and a body of booleans plus model ids. Widening
1411
+ // `/api/*` or handing the client an admin token to read `GET /api/providers` would
1412
+ // have traded a reporting defect for a credential one.
1413
+ //
1414
+ // What it discloses beyond /v1/catalog and /v1/models, exactly: `hasCredential`,
1415
+ // `loggedIn`, `authMode`, the featured roster, and the NAME and adapter of an ENABLED
1416
+ // provider those routes omit for want of a usable credential — which is the point of
1417
+ // the route. A `disabled` provider is NOT exported (`buildHubState` drops it), because
1418
+ // the catalog filters it out too and naming it here would be the only place a data key
1419
+ // learns of it.
1420
+ //
1421
+ // Placed between /v1/catalog and /v1/models so all three least-privilege client reads
1422
+ // stay in sight of each other.
1423
+ const admission = resolveApiAuth(req, policy);
1424
+ if (!admission) return withCors(formatErrorResponse(401, "authentication_error", "opencodex API key required"), req, policy);
1425
+ if (!isAllowedRequestOrigin(req, policy)) {
1426
+ return withCors(formatErrorResponse(403, "origin_rejected", "cross-origin data-plane request blocked"), req, policy);
1427
+ }
1428
+ // Role gate AFTER admission, deliberately: answering an unauthenticated caller would
1429
+ // turn this into a free "is that machine a hub?" probe. A standalone or client install
1430
+ // gains no surface at all — the route simply does not exist there.
1431
+ //
1432
+ // Built, not formatErrorResponse'd, for the same reason /v1/catalog builds its 404: the
1433
+ // code has to distinguish "this route exists and this host is not a hub" from "this
1434
+ // build has no such route", which is the difference between admission proof and a
1435
+ // vacuous pass in tests/server/api-key-attribution.test.ts.
1436
+ if (config.runtimeRole !== "hub") {
1437
+ return withCors(
1438
+ new Response(JSON.stringify({
1439
+ error: {
1440
+ type: "invalid_request_error",
1441
+ code: "hub_state_not_a_hub",
1442
+ message: "hub state is served only by a host whose runtimeRole is hub",
1443
+ },
1444
+ }), { status: 404, headers: { "content-type": "application/json" } }),
1445
+ req,
1446
+ policy,
1447
+ );
1448
+ }
1449
+ const { buildHubState } = await import("./hub-state");
1450
+ const { MAX_HUB_STATE_BYTES } = await import("../remote/hub-state");
1451
+ const { oauthLoginSummary } = await import("../oauth");
1452
+ // `true` masks emails, but the projection drops the field entirely; passing the mask
1453
+ // anyway means a future refactor that starts copying fields cannot leak a raw address.
1454
+ const body = JSON.stringify(buildHubState(config, oauthLoginSummary(true), VERSION));
1455
+ const bytes = Buffer.byteLength(body);
1456
+ if (bytes > MAX_HUB_STATE_BYTES) {
1457
+ return withCors(
1458
+ new Response(JSON.stringify({
1459
+ error: { type: "server_error", code: "hub_state_too_large", message: "hub state exceeds the maximum served size" },
1460
+ }), { status: 507, headers: { "content-type": "application/json" } }),
1461
+ req,
1462
+ policy,
1463
+ );
1464
+ }
1465
+ return withCors(
1466
+ new Response(req.method === "HEAD" ? null : body, {
1467
+ status: 200,
1468
+ headers: {
1469
+ "content-type": "application/json",
1470
+ // Varies by credential-bearing identity and by live login state: never cached,
1471
+ // and no validator to revalidate with (same rule as /v1/catalog).
1472
+ "cache-control": "no-store",
1473
+ "content-length": String(bytes),
1474
+ },
1475
+ }),
1476
+ req,
1477
+ policy,
1478
+ );
1479
+ }
1480
+
1350
1481
  if (url.pathname === "/v1/models" && req.method === "GET") {
1351
1482
  // #809: the catalog read sits immediately before model discovery because it shares
1352
1483
  // that route's admission rationale exactly. Keep them adjacent so a future change to
@@ -1981,10 +2112,13 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
1981
2112
  ...admissionFields(admission),
1982
2113
  inboundProtocol: "chat",
1983
2114
  };
2115
+ // `policy`, not `config`: this route is now served on the unauthenticated loopback
2116
+ // listener too (#4236), and only the receiving listener's view produces CORS headers
2117
+ // that match the admission decision made above.
1984
2118
  return runAdmittedHttpTurn(req, policy, async turnAdmissionLease => withCors(
1985
2119
  await handleChatCompletions(req, config, logCtx, { requestId, start, turnAdmissionLease, admission }),
1986
2120
  req,
1987
- config,
2121
+ policy,
1988
2122
  ));
1989
2123
  }
1990
2124
 
@@ -2488,10 +2622,17 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
2488
2622
  // who forgot, has to be able to see that an unauthenticated surface is live without
2489
2623
  // reading the file.
2490
2624
  const loopbackPort = loopbackServer.port ?? loopbackListenerPort;
2491
- console.warn(`⚠️ Unauthenticated loopback listener active on http://127.0.0.1:${loopbackPort}`);
2492
- console.warn(` Any local process can use it without a credential — it spends account`);
2493
- console.warn(` quota and paid provider credentials, and can starve authenticated`);
2494
- console.warn(` remote clients. Not for shared or multi-tenant hosts.`);
2625
+ if (loopbackListener?.enabled === true && loopbackListener.port === undefined) {
2626
+ // The companion form is the intended one-port hub topology, not a surprise surface: the
2627
+ // public listener is already on a non-loopback address, so this line states where local
2628
+ // processes go rather than warning about a second port nobody asked for.
2629
+ console.log(`🔁 Loopback companion active on http://127.0.0.1:${loopbackPort} — same port as the public listener; local processes need no credential`);
2630
+ } else {
2631
+ console.warn(`⚠️ Unauthenticated loopback listener active on http://127.0.0.1:${loopbackPort}`);
2632
+ console.warn(` Any local process can use it without a credential — it spends account`);
2633
+ console.warn(` quota and paid provider credentials, and can starve authenticated`);
2634
+ console.warn(` remote clients. Not for shared or multi-tenant hosts.`);
2635
+ }
2495
2636
  }
2496
2637
 
2497
2638
  if (managementIngressServer) {
@@ -1,4 +1,6 @@
1
1
  import type { OcxConfig } from "../../types";
2
+ import { isWildcardHostname } from "../../codex/loopback-target";
3
+ import { localInferenceDestination } from "../../lib/local-destinations";
2
4
  import { probeHostname } from "../proxy-liveness";
3
5
 
4
6
  export interface ApiAccessEndpoints {
@@ -21,9 +23,14 @@ export type BuildApiAccessEndpointsOptions = {
21
23
  requestOrigin?: string | null;
22
24
  };
23
25
 
26
+ /**
27
+ * Wildcard bind scope, shared with `probeHostname` and the loopback-companion gate rather than
28
+ * re-spelled here: a third list of three spellings is how `0.0.0.0.` and `::0` ended up treated
29
+ * as specific bind addresses on one side and wildcards on the other.
30
+ */
24
31
  function isWildcardBindHost(hostname: string | undefined): boolean {
25
32
  const trimmed = (hostname ?? "").trim();
26
- return !trimmed || trimmed === "0.0.0.0" || trimmed === "::" || trimmed === "[::]";
33
+ return !trimmed || isWildcardHostname(trimmed);
27
34
  }
28
35
 
29
36
  /** Bracket bare IPv6 literals for URL authority composition. */
@@ -66,7 +73,7 @@ function originBaseUrl(raw: string): string | null {
66
73
  * Falls back to loopback only when no usable request context is available.
67
74
  */
68
75
  export function resolveApiAccessBaseUrl(
69
- config: Pick<OcxConfig, "hostname" | "port">,
76
+ config: Pick<OcxConfig, "hostname" | "port" | "unauthenticatedLoopbackListener">,
70
77
  opts: BuildApiAccessEndpointsOptions = {},
71
78
  ): string {
72
79
  const port = config.port ?? 10100;
@@ -104,7 +111,11 @@ export function resolveApiAccessBaseUrl(
104
111
  }
105
112
  }
106
113
 
107
- return `http://127.0.0.1:${port}/v1`;
114
+ // Last resort: a wildcard bind with no usable request context, so the only address we can
115
+ // name is loopback — and on that address the unauthenticated loopback listener, when one is
116
+ // enabled, is the port a local caller should use (#4236). The branches above are unchanged:
117
+ // a specific bind or a real request host still describes the address the CLIENT reached.
118
+ return `${localInferenceDestination(config, port).origin}/v1`;
108
119
  }
109
120
 
110
121
  /** @deprecated Prefer resolveApiAccessBaseUrl; retained for focused host-format tests. */
@@ -110,7 +110,7 @@ import { applySystemEnvToggle } from "../system-env";
110
110
  import { getCachedStartupHealth, invalidateStartupHealthCache } from "../startup-health-cache";
111
111
  import { runWindowsTrayAction } from "../windows-tray-control";
112
112
  import { runStartupInstallAction, type StartupInstallAction } from "../startup-action-control";
113
- import { displayCodexRuntimePath, effortClampAppliesToRuntime, loadLastEffortClamp, resolveCodexRuntime } from "../../codex/runtime";
113
+ import { displayCodexRuntimePath, effortClampAppliesToRuntime, liveRemovedEfforts, loadLastEffortClamp, resolveCodexRuntime } from "../../codex/runtime";
114
114
 
115
115
  import { isPlainRecord, parseDebugLogQuery, tokPerSecondResult, unavailableCostReason, costResult, requestLogDto, stripRegistryOnlyStaticHeaders, fetchAllModels } from "./shared";
116
116
  import type { MetricUnavailableReason, TokPerSecondResult, CostEstimateReason, CostResult, MetricSource } from "./shared";
@@ -344,7 +344,7 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise<Respon
344
344
  : null,
345
345
  catalogClamp: {
346
346
  active: clampActive,
347
- removedEfforts: clampActive ? (lastClamp?.removedEfforts ?? []) : [],
347
+ removedEfforts: clampActive ? [...liveRemovedEfforts(lastClamp)] : [],
348
348
  runtimeVersion: clampActive ? (lastClamp?.runtimeVersion ?? null) : null,
349
349
  },
350
350
  warning: warningParts.length > 0 ? warningParts.join(" ") : null,
@@ -14,6 +14,7 @@ import { cursorLastSeen, type CursorSeen } from "../../integrations/cursor-seen"
14
14
  import { detectCursorInstalls, type CursorInstall } from "../../integrations/cursor-detect";
15
15
  import { loadCursorEffortTable } from "../../integrations/cursor-effort-table";
16
16
  import { configuredApiAuthToken, isApiAuthRequired, jsonResponse } from "../auth-cors";
17
+ import { localInferenceDestination } from "../../lib/local-destinations";
17
18
  import { fetchAllModels } from "../management-api";
18
19
  import { predictCursorEffort } from "../models-capabilities";
19
20
  import { expandCursorEffortRow, knownEffortRowIds } from "../effort-row";
@@ -54,11 +55,19 @@ export async function buildCursorIntegrationStatus(
54
55
  // The port the browser reached is the one Cursor on the same machine will reach too; the
55
56
  // runtime record and config.port are fallbacks for a request that carries no port.
56
57
  const port = runtime?.port ?? (Number(ctx.url?.port) || config.port);
57
- // Describes the public bind. A second unauthenticated loopback listener may exist, but the
58
- // value a user pastes into Cursor must work against the bind they will actually reach.
58
+ // Cursor runs on this machine, so the gateway URL it is told to paste is the LOCAL one: the
59
+ // unauthenticated loopback listener when one is enabled, and otherwise the bind address on the
60
+ // public port — 127.0.0.1 for a loopback or wildcard bind exactly as before, and the tailnet
61
+ // or LAN address on a hub, where no loopback socket exists to paste (#4236).
62
+ const gateway = localInferenceDestination(config, port ?? 10100);
63
+ // apiKeyMode describes the admission rule of the destination just resolved, which on the
64
+ // loopback listener is "no key needed" and on every other form is "a key is required".
65
+ // Pasting one into the listener is harmless; omitting one on a bind that demands it is not.
59
66
  const credentialConfigured = !!configuredApiAuthToken(config)
60
67
  || (config.apiKeys ?? []).some(entry => !!entry.key.trim());
61
- const apiKeyMode = isApiAuthRequired(config) || credentialConfigured ? "credential" : "placeholder";
68
+ const apiKeyMode = gateway.requiresAdmissionToken || isApiAuthRequired(config) || credentialConfigured
69
+ ? "credential"
70
+ : "placeholder";
62
71
 
63
72
  const limits = nativeContextLimits(config);
64
73
  // Same visibility rules as the raw /v1/models list Cursor will read: disabled models and
@@ -107,7 +116,7 @@ export async function buildCursorIntegrationStatus(
107
116
  },
108
117
  regularCursor: { installed: regular !== undefined, path: regular?.path ?? null },
109
118
  gateway: {
110
- baseUrl: `http://127.0.0.1:${port}/v1`,
119
+ baseUrl: `${gateway.origin}/v1`,
111
120
  apiKeyMode,
112
121
  placeholder: CURSOR_GATEWAY_PLACEHOLDER_KEY,
113
122
  },
@@ -115,7 +115,10 @@ export async function handleLogsUsageRoutes(ctx: ManagementContext): Promise<Res
115
115
  }
116
116
  const all = getRequestLogEntries();
117
117
  const total = filteredRequestLogCount(all, url.searchParams);
118
- const logs = filterRequestLogs(all, url.searchParams).map(requestLogDto);
118
+ // Not point-free: requestLogDto takes an options object second, and Array.map would pass the
119
+ // element INDEX into it. An explicit arrow keeps the default (decode rate included) and is
120
+ // what /api/logs wants; /api/request-history opts out at its own call sites.
121
+ const logs = filterRequestLogs(all, url.searchParams).map(entry => requestLogDto(entry));
119
122
  const poll = selectRequestLogPoll(logs, url.searchParams, cursor);
120
123
  return jsonResponse({
121
124
  timeZone: Intl.DateTimeFormat().resolvedOptions().timeZone,
@@ -147,10 +147,25 @@ export async function listManagementModelRows(
147
147
  };
148
148
  });
149
149
  const publicModels = uniqueCatalogModelsForPublicList(models);
150
+ // Custom rows below are REBUILT from config.customModels rather than spread from a
151
+ // CatalogModel, so every field gather computed for the same slug has to be carried across by
152
+ // hand. Without this a custom model whose provider is out of credit would be the one row on
153
+ // the page that never shows as inactive (#1711), because the gather-derived row it replaces
154
+ // is dropped by the slug dedup below.
155
+ const quotaInactiveByNamespaced = new Map(
156
+ publicModels
157
+ .filter(model => model.quotaInactiveReason !== undefined)
158
+ .map(model => [catalogModelSlug(model), model.quotaInactiveReason!] as const),
159
+ );
150
160
  const comboNamespaced = new Set(
151
161
  publicModels.filter(model => model.provider === "combo").map(catalogModelSlug),
152
162
  );
153
- const visibleCustomModels = customModels.filter(model => !comboNamespaced.has(model.namespaced));
163
+ const visibleCustomModels = customModels
164
+ .filter(model => !comboNamespaced.has(model.namespaced))
165
+ .map(model => {
166
+ const quotaInactiveReason = quotaInactiveByNamespaced.get(model.namespaced);
167
+ return quotaInactiveReason ? { ...model, quotaInactiveReason } : model;
168
+ });
154
169
  // Custom metadata wins when a physical live/static row resolves to the same Codex-facing
155
170
  // slug, while a combo keeps the same precedence it has in routing and /v1/models.
156
171
  const customNamespaced = new Set(visibleCustomModels.map(c => c.namespaced));
@@ -25,6 +25,7 @@ import {
25
25
  } from "../../oauth";
26
26
  import { OAuthMutationBusyError, removeCredential } from "../../oauth/store";
27
27
  import { providerDestinationResolvedError } from "../../lib/destination-policy";
28
+ import { emailMaskingEnabled } from "../../lib/privacy";
28
29
  import { reconcileLiveStateStores } from "../../lib/state-store-registrations";
29
30
  import { enrichProviderFromCatalog, listKeyLoginProviders } from "../../oauth/key-providers";
30
31
  import { deriveProviderPresets } from "../../providers/derive";
@@ -233,7 +234,10 @@ export async function handleOauthAccountRoutes(ctx: ManagementContext): Promise<
233
234
  if (url.pathname === "/api/oauth/status" && req.method === "GET") {
234
235
  const provider = (url.searchParams.get("provider") ?? "").trim().toLowerCase();
235
236
  if (!isPublicOAuthProvider(provider)) return jsonResponse({ error: "unknown oauth provider" }, 400);
236
- const status = getLoginStatus(provider);
237
+ // Resolved here, at the request boundary that already holds the config, and passed down.
238
+ // getLoginStatus stays free of config I/O. This route does not re-mask afterwards: it
239
+ // consumes the already-projected status rather than redacting a second time.
240
+ const status = getLoginStatus(provider, emailMaskingEnabled(config));
237
241
  return jsonResponse(status);
238
242
  }
239
243
 
@@ -269,7 +273,7 @@ export async function handleOauthAccountRoutes(ctx: ManagementContext): Promise<
269
273
  } = await import("../../oauth/health");
270
274
  const projectAccounts = () => {
271
275
  const set = getAccountSet(provider);
272
- const current = getLoginStatus(provider);
276
+ const current = getLoginStatus(provider, emailMaskingEnabled(config));
273
277
  return {
274
278
  activeAccountId: current.activeAccountId ?? null,
275
279
  accounts: (current.accounts ?? []).map(summary => {
@@ -54,6 +54,7 @@ import {
54
54
  readBoundedDiscoveryJson,
55
55
  resolveProviderModelDiscovery,
56
56
  } from "../../providers/model-discovery";
57
+ import { extractGoogleAiStudioModelItems } from "../../providers/google-ai-studio-model-discovery";
57
58
  import { routedSlug, slugEquals } from "../../providers/slug-codec";
58
59
  import { clearAccountQuotaCache, clearProviderQuotaCache, fetchProviderQuotaReports } from "../../providers/quota";
59
60
  import { clearKeyCooldowns } from "../../providers/key-failover";
@@ -1428,13 +1429,19 @@ export async function handleProviderRoutes(ctx: ManagementContext): Promise<Resp
1428
1429
  return jsonResponse({ ok: false, latencyMs, error: "upstream CCA model discovery returned an unexpected shape" });
1429
1430
  }
1430
1431
  // OpenAI-style lists (and Together top-level arrays) use the same validation/dedupe/filter
1431
- // as catalog discovery. Google's /v1beta/models uses `models[].name` and remains a
1432
- // connectivity-only count because it is not an authoritative catalog source.
1432
+ // as catalog discovery. Google AI Studio parses the native `models[]` envelope and filters to
1433
+ // `generateContent`, while other providers fall back to generic envelope rows if they return `models[]`.
1433
1434
  const record = bounded.value !== null && typeof bounded.value === "object" && !Array.isArray(bounded.value)
1434
1435
  ? bounded.value as Record<string, unknown>
1435
1436
  : undefined;
1437
+ const isAiStudio = effectiveGoogleMode(name, prov) === "ai-studio";
1438
+ const googleAiStudio = !ccaModels && isAiStudio
1439
+ ? extractGoogleAiStudioModelItems(bounded.value, discovery.maxModels)
1440
+ : undefined;
1436
1441
  const extracted = ccaModels
1437
1442
  ? undefined
1443
+ : googleAiStudio?.ok
1444
+ ? googleAiStudio
1438
1445
  : Array.isArray(bounded.value) || Array.isArray(record?.data)
1439
1446
  ? extractProviderModelItems(bounded.value, discovery)
1440
1447
  : extractModelEnvelopeRows(bounded.value, discovery.maxModels, ["models"]);
@@ -106,7 +106,9 @@ export async function handleRequestHistoryRoutes(ctx: ManagementContext): Promis
106
106
  to,
107
107
  }, cursor, limit);
108
108
  return jsonResponse({
109
- entries: page.rows.map(row => requestLogDto(requestLogEntryFromPersistedUsage(row))),
109
+ // The decode rate is a Logs-page metric; this endpoint shares the DTO but not its
110
+ // contract, so it opts out rather than silently widening its own response shape (#4038).
111
+ entries: page.rows.map(row => requestLogDto(requestLogEntryFromPersistedUsage(row), { includeDecodeRate: false })),
110
112
  ...(page.nextCursor ? { nextCursor: page.nextCursor } : {}),
111
113
  hasMore: page.hasMore,
112
114
  index: {
@@ -184,7 +186,7 @@ export async function handleRequestHistoryRoutes(ctx: ManagementContext): Promis
184
186
  if (!entry) {
185
187
  return jsonResponse({ error: { code: "not_found", message: "unknown request" } }, 404, req, config);
186
188
  }
187
- return jsonResponse(requestLogDto(requestLogEntryFromPersistedUsage(entry)), 200, req, config);
189
+ return jsonResponse(requestLogDto(requestLogEntryFromPersistedUsage(entry), { includeDecodeRate: false }), 200, req, config);
188
190
  }
189
191
 
190
192
  return null;
@@ -95,6 +95,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
95
95
  { method: "PATCH", path: "/api/codex-auth/pool-strategy", module: "codex/auth-api", mutates: true },
96
96
  { method: "POST", path: "/api/codex-auth/accounts", module: "codex/auth-api", mutates: true },
97
97
  { method: "POST", path: "/api/codex-auth/accounts/clear-cooldown", module: "codex/auth-api", mutates: true },
98
+ { method: "POST", path: "/api/codex-auth/accounts/refresh", module: "codex/auth-api", mutates: true },
98
99
  { method: "POST", path: "/api/codex-auth/login", module: "codex/auth-api", mutates: true },
99
100
  { method: "POST", path: "/api/codex-auth/login/cancel", module: "codex/auth-api", mutates: true },
100
101
  { method: "POST", path: "/api/codex-auth/login/code", module: "codex/auth-api", mutates: true },
@@ -148,9 +149,9 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
148
149
  { method: "GET", path: "/api/client-integrations/aside/profiles/{profileId}", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
149
150
  { method: "PUT", path: "/api/client-integrations/aside/profiles/{profileId}", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode" },
150
151
  { method: "GET", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
151
- { method: "DELETE", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside history deletion uses the dashboard journal cleanup; the CLI has history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_plan/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
152
+ { method: "DELETE", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside history deletion uses the dashboard journal cleanup; the CLI has history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
152
153
  { method: "GET", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
153
- { method: "DELETE", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside profile history deletion uses the dashboard journal cleanup; the CLI has scoped history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_plan/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
154
+ { method: "DELETE", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside profile history deletion uses the dashboard journal cleanup; the CLI has scoped history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
154
155
  { method: "POST", path: "/api/client-integrations/aside/profiles/{profileId}/restore", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode" },
155
156
  // server/management/codex-prompt-routes
156
157
  { method: "GET", path: "/api/codex-prompt", module: "server/management/codex-prompt-routes", mutates: false },
@@ -186,7 +187,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
186
187
  // server/management/integration-routes
187
188
  { method: "GET", path: "/api/client-integrations", module: "server/management/integration-routes", mutates: false },
188
189
  { method: "GET", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: false },
189
- { method: "DELETE", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Retiring one rollback row is a dashboard-local cleanup; the CLI verb that would drive it is owed by a later work-phase and is not implemented here.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_plan/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
190
+ { method: "DELETE", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Retiring one rollback row is a dashboard-local cleanup; the CLI verb that would drive it is owed by a later work-phase and is not implemented here.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
190
191
  { method: "POST", path: "/api/client-integrations/restore", module: "server/management/integration-routes", mutates: true },
191
192
  // server/management/lab-automation-routes
192
193
  { method: "GET", path: "/api/lab/automation", module: "server/management/lab-automation-routes", mutates: false, exempt: { reason: "local-transport", why: "ocx lab reads the same rows from the local SQLite projection; src/cli/lab.ts imports ../lab/query directly and never fetches /api/lab." } },
@@ -292,7 +293,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
292
293
  { method: "PATCH", path: "/api/providers", module: "server/management/provider-routes", mutates: true },
293
294
  { method: "POST", path: "/api/providers", module: "server/management/provider-routes", mutates: true },
294
295
  { method: "POST", path: "/api/providers/test", module: "server/management/provider-routes", mutates: true },
295
- { method: "PUT", path: "/api/providers", module: "server/management/provider-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Issue #3280 scopes this atomic batch endpoint to the GUI JSON editor; a matching CLI verb is outside wp5 and remains owed.", owner: "wp5-followup", ownerDoc: "devlog/_plan/260903_bug_drawdown_bcda/050_phase5.md" } },
296
+ { method: "PUT", path: "/api/providers", module: "server/management/provider-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Issue #3280 scopes this atomic batch endpoint to the GUI JSON editor; a matching CLI verb is outside wp5 and remains owed.", owner: "wp5-followup", ownerDoc: "devlog/_fin/260903_bug_drawdown_bcda/050_phase5.md" } },
296
297
  { method: "PUT", path: "/api/provider-context-caps", module: "server/management/provider-routes", mutates: true },
297
298
  // server/management/quota-reset-routes
298
299
  { method: "GET", path: "/api/quota-resets", module: "server/management/quota-reset-routes", mutates: false, mechanism: "negated-guard" },
@@ -79,7 +79,8 @@ export function parseDebugLogQuery(url: URL): { after: number; limit: number } {
79
79
  export type MetricUnavailableReason =
80
80
  | "usage_missing" | "usage_unsupported" | "output_missing" | "invalid_duration"
81
81
  | "price_unmatched" | "invalid_cache_breakdown"
82
- | "invalid_usage" | "combo_attempt_unavailable";
82
+ | "invalid_usage" | "combo_attempt_unavailable"
83
+ | "ttft_missing" | "decode_window_too_short";
83
84
 
84
85
  export type TokPerSecondResult =
85
86
  | { kind: "value"; value: number; estimated: boolean }
@@ -96,7 +97,7 @@ export type CostResult =
96
97
  | { kind: "value"; estimate: NonNullable<ReturnType<typeof estimateRequestCost>>; estimateReasons: CostEstimateReason[] }
97
98
  | { kind: "unavailable"; reason: MetricUnavailableReason };
98
99
 
99
- export type MetricSource = Pick<RequestLogEntry, "provider" | "model" | "durationMs" | "usageStatus" | "usage" | "requestedServiceTier" | "configuredServiceTier" | "responseServiceTier" | "tierOutcome" | "routeDecision"> & {
100
+ export type MetricSource = Pick<RequestLogEntry, "provider" | "model" | "durationMs" | "firstOutputMs" | "usageStatus" | "usage" | "requestedServiceTier" | "configuredServiceTier" | "responseServiceTier" | "tierOutcome" | "routeDecision"> & {
100
101
  attempts?: readonly PersistedUsageAttempt[];
101
102
  };
102
103
 
@@ -113,6 +114,51 @@ export function tokPerSecondResult(entry: Pick<MetricSource, "durationMs" | "usa
113
114
  return { kind: "value", value, estimated: entry.usageStatus === "estimated" || entry.usage.estimated === true };
114
115
  }
115
116
 
117
+ /**
118
+ * Shortest post-TTFT window that can carry a decode-rate estimate (#4038).
119
+ *
120
+ * Below one second the window is dominated by things that are not decoding: TTFT jitter, the
121
+ * proxy's own buffering, and the granularity of the timestamps themselves. A 240-token response
122
+ * whose first token arrived 50 ms before the last one is not a 4800 tok/s model, and printing
123
+ * that number is worse than printing nothing — which is precisely why the earlier attempt at
124
+ * this metric (#4040) was closed as an unreliable estimate.
125
+ */
126
+ export const MIN_DECODE_WINDOW_MS = 1_000;
127
+
128
+ /**
129
+ * Estimated DECODE throughput: output tokens over the window after the first token (#4038).
130
+ *
131
+ * Strictly additive. `tokensPerSecond`, `tokPerSecondResult`, `RequestLogEntry` and
132
+ * `usage.jsonl` are untouched, and the end-to-end rate beside it keeps meaning exactly what it
133
+ * has always meant — it is documented as end-to-end, so this is a missing metric rather than a
134
+ * miscalculated one.
135
+ *
136
+ * Always `estimated: true`. The proxy's TTFT is when the FIRST BYTE reached the proxy, which is
137
+ * not the provider's own generation start, so this can never be more than an estimate no matter
138
+ * how long the window is. Saying so in the payload is the honest half of the answer to #4040;
139
+ * MIN_DECODE_WINDOW_MS is the other half.
140
+ */
141
+ export function decodeTokPerSecondResult(
142
+ entry: Pick<MetricSource, "durationMs" | "firstOutputMs" | "usageStatus" | "usage">,
143
+ ): TokPerSecondResult {
144
+ if (!entry.usage) return { kind: "unavailable", reason: "usage_missing" };
145
+ if (entry.usageStatus === "unsupported") return { kind: "unavailable", reason: "usage_unsupported" };
146
+ if (entry.usage.outputTokens <= 0) return { kind: "unavailable", reason: "output_missing" };
147
+ // A row that predates TTFT capture, or a non-streaming turn that never recorded one, has no
148
+ // window to measure. That is a different fact from a bad duration, so it gets its own reason.
149
+ if (entry.firstOutputMs === undefined) return { kind: "unavailable", reason: "ttft_missing" };
150
+ if (!Number.isFinite(entry.firstOutputMs) || entry.firstOutputMs < 0 || !Number.isFinite(entry.durationMs)) {
151
+ return { kind: "unavailable", reason: "invalid_duration" };
152
+ }
153
+ const windowMs = entry.durationMs - entry.firstOutputMs;
154
+ // TTFT at or past the total duration means the two clocks disagree; there is no window.
155
+ if (windowMs <= 0) return { kind: "unavailable", reason: "invalid_duration" };
156
+ if (windowMs < MIN_DECODE_WINDOW_MS) return { kind: "unavailable", reason: "decode_window_too_short" };
157
+ const value = tokensPerSecond(entry.usage.outputTokens, windowMs);
158
+ if (value === null) return { kind: "unavailable", reason: "invalid_duration" };
159
+ return { kind: "value", value, estimated: true };
160
+ }
161
+
116
162
  export function unavailableCostReason(entry: MetricSource): MetricUnavailableReason {
117
163
  // Normalizer-first classification: the landed normalizer recovers legacy
118
164
  // cachedInputTokens=read+write rows via retry, so a raw read+write>input
@@ -152,11 +198,26 @@ export function costResult(entry: MetricSource): CostResult {
152
198
  return { kind: "value", estimate, estimateReasons };
153
199
  }
154
200
 
155
- export function requestLogDto(entry: RequestLogEntry): Record<string, unknown> {
201
+ /**
202
+ * `/api/logs` row projection.
203
+ *
204
+ * `includeDecodeRate` exists because `/api/request-history` shares this DTO but not its
205
+ * contract (#4038). The value would be meaningful there — `firstOutputMs` does survive into a
206
+ * persisted-usage row — so this is a scope decision, not a correctness one: the decode rate is
207
+ * a Logs-page metric, and widening a separate endpoint's response shape is not this change's
208
+ * business. Flipping it on later is one argument.
209
+ */
210
+ export function requestLogDto(
211
+ entry: RequestLogEntry,
212
+ { includeDecodeRate = true }: { includeDecodeRate?: boolean } = {},
213
+ ): Record<string, unknown> {
156
214
  return {
157
215
  ...entry,
158
216
  displayMetrics: {
159
217
  tokPerSecond: tokPerSecondResult(entry),
218
+ // The parent uses the REQUEST's own TTFT. A combo parent must not borrow an attempt's,
219
+ // which would measure a window the parent never had.
220
+ ...(includeDecodeRate ? { decodeTokPerSecond: decodeTokPerSecondResult(entry) } : {}),
160
221
  cost: costResult(entry),
161
222
  },
162
223
  ...(entry.attempts?.length
@@ -165,6 +226,8 @@ export function requestLogDto(entry: RequestLogEntry): Record<string, unknown> {
165
226
  ...attempt,
166
227
  displayMetrics: {
167
228
  tokPerSecond: tokPerSecondResult(attempt),
229
+ // Each attempt measures its own attempt-relative TTFT.
230
+ ...(includeDecodeRate ? { decodeTokPerSecond: decodeTokPerSecondResult(attempt) } : {}),
168
231
  cost: costResult({ ...attempt, attempts: undefined, routeDecision: entry.routeDecision, requestedServiceTier: entry.requestedServiceTier, configuredServiceTier: entry.configuredServiceTier, responseServiceTier: entry.responseServiceTier }),
169
232
  },
170
233
  })),
@@ -385,7 +385,7 @@ export async function handleManagementAPI(
385
385
  const { ConfigMutationLockError } = await import("../config");
386
386
  const { CodexCredentialRefreshLockTimeoutError } = await import("../codex/account-store");
387
387
  try {
388
- return await handleCodexAuthAPI(req, url, config, convergeCodexCatalog);
388
+ return await handleCodexAuthAPI(req, url, config, convergeCodexCatalog, principal);
389
389
  } catch (error) {
390
390
  // Credential writers remap ConfigMutationLockError to CodexCredentialRefreshLockTimeoutError;
391
391
  // treat both as the same retryable busy response.