@bitkyc08/opencodex 2.49.0 → 2.51.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS_INSTALL.md +9 -1
- package/README.md +3 -0
- package/bin/ocx.mjs +222 -71
- package/gui/dist/assets/index-D7BdZpZm.js +115 -0
- package/gui/dist/index.html +1 -1
- package/package.json +1 -1
- package/src/adapters/qoder/adapter.ts +69 -1
- package/src/adapters/qoder/scaffold-guard.ts +233 -0
- package/src/claude/agents-inject.ts +29 -5
- package/src/claude/desktop-3p.ts +31 -3
- package/src/claude/gateway-cache.ts +12 -21
- package/src/claude/inbound.ts +17 -5
- package/src/cli/account-api.ts +18 -3
- package/src/cli/account-auth.ts +8 -1
- package/src/cli/account-extended.ts +2 -1
- package/src/cli/account.ts +1 -0
- package/src/cli/capabilities.ts +43 -1
- package/src/cli/claude-agent-startup-sync.ts +26 -1
- package/src/cli/claude.ts +138 -20
- package/src/cli/config-command.ts +67 -1
- package/src/cli/connect.ts +181 -14
- package/src/cli/dispatch.ts +53 -9
- package/src/cli/doctor.ts +9 -2
- package/src/cli/ensure-desired-integrations.ts +10 -0
- package/src/cli/gui-pair-client.ts +1 -12
- package/src/cli/help.ts +4 -1
- package/src/cli/hub.ts +367 -0
- package/src/cli/index.ts +99 -31
- package/src/cli/launcher-context.ts +1 -1
- package/src/cli/models-runtime.ts +8 -3
- package/src/cli/observe.ts +13 -3
- package/src/cli/registry.ts +43 -3
- package/src/cli/status.ts +325 -5
- package/src/cli/version-skew.ts +4 -1
- package/src/cli.ts +2 -2
- package/src/client/catalog-compatibility.ts +192 -0
- package/src/client/connect.ts +31 -0
- package/src/client/hub-client.ts +52 -0
- package/src/client/hub-state.ts +214 -0
- package/src/clients/config-export/zcode.ts +24 -0
- package/src/codex/account-runtime-state.ts +6 -1
- package/src/codex/account-store.ts +72 -9
- package/src/codex/account-usability.ts +50 -13
- package/src/codex/auth-api.ts +156 -28
- package/src/codex/auth-context.ts +21 -0
- package/src/codex/catalog/effort.ts +67 -8
- package/src/codex/catalog/parsing.ts +23 -0
- package/src/codex/catalog/provider-fetch.ts +71 -2
- package/src/codex/catalog/sync.ts +99 -0
- package/src/codex/codex-write-lock.ts +11 -2
- package/src/codex/desired-state.ts +47 -1
- package/src/codex/inject-coordination.ts +10 -5
- package/src/codex/inject.ts +29 -12
- package/src/codex/loopback-target.ts +45 -0
- package/src/codex/quota-auto-refresh.ts +6 -1
- package/src/codex/quota.ts +54 -8
- package/src/codex/routing.ts +48 -1
- package/src/codex/runtime.ts +37 -3
- package/src/codex/sync.ts +29 -9
- package/src/codex/warmup.ts +21 -4
- package/src/combos/index.ts +2 -0
- package/src/combos/resolve.ts +52 -0
- package/src/config/pending-teardown.ts +1 -1
- package/src/config.ts +184 -12
- package/src/generated/compatibility-version.json +188 -116
- package/src/grok/status.ts +9 -1
- package/src/integrations/config-io.ts +54 -1
- package/src/lib/bun-runtime.ts +1 -1
- package/src/lib/errors.ts +8 -0
- package/src/lib/gui-pair-capability.ts +27 -0
- package/src/lib/local-destinations.ts +162 -0
- package/src/lib/package-tree-integrity.ts +1 -1
- package/src/lib/privacy.ts +25 -0
- package/src/lib/process-control.ts +130 -20
- package/src/lib/service-secrets.ts +28 -0
- package/src/lib/test-home-guard.ts +49 -0
- package/src/oauth/health.ts +47 -12
- package/src/oauth/index.ts +46 -8
- package/src/oauth/token-guardian.ts +32 -6
- package/src/providers/google-ai-studio-model-discovery.ts +74 -0
- package/src/providers/opencode-go-transport.ts +9 -1
- package/src/providers/opencode-zen-rate-limit.ts +75 -0
- package/src/providers/quota.ts +20 -1
- package/src/providers/registry.ts +35 -6
- package/src/remote/hub-state.ts +182 -0
- package/src/server/auth-cors.ts +11 -0
- package/src/server/chat-completions.ts +10 -7
- package/src/server/chat-native.ts +10 -1
- package/src/server/claude-messages.ts +12 -6
- package/src/server/hub-state.ts +98 -0
- package/src/server/images.ts +2 -2
- package/src/server/index.ts +149 -8
- package/src/server/management/api-access.ts +14 -3
- package/src/server/management/config-routes.ts +2 -2
- package/src/server/management/cursor-integration-routes.ts +13 -4
- package/src/server/management/logs-usage-routes.ts +4 -1
- package/src/server/management/model-rows.ts +16 -1
- package/src/server/management/oauth-account-routes.ts +6 -2
- package/src/server/management/provider-routes.ts +9 -2
- package/src/server/management/request-history-routes.ts +4 -2
- package/src/server/management/route-registry.ts +5 -4
- package/src/server/management/shared.ts +66 -3
- package/src/server/management-api.ts +1 -1
- package/src/server/proxy-liveness.ts +7 -1
- package/src/server/request-decompress.ts +91 -3
- package/src/server/request-log-conversation.ts +41 -1
- package/src/server/request-log.ts +10 -0
- package/src/server/responses/codex-auth-error.ts +18 -1
- package/src/server/responses/codex-ws-exchange.ts +36 -4
- package/src/server/responses/codex-ws-wire.ts +76 -5
- package/src/server/responses/compact.ts +28 -11
- package/src/server/responses/context-overflow.ts +11 -0
- package/src/server/responses/core.ts +201 -48
- package/src/server/responses/policy-fallback.ts +13 -3
- package/src/server/search.ts +2 -2
- package/src/server/system-env-shell.ts +14 -2
- package/src/server/system-env.ts +106 -14
- package/src/service.ts +965 -68
- package/src/types/accounts.ts +18 -0
- package/src/types/config.ts +93 -4
- package/src/types/provider.ts +56 -0
- package/src/types.ts +4 -0
- package/src/update/badge.ts +3 -2
- package/src/update/index.ts +317 -64
- package/src/update/install-detection.d.mts +6 -0
- package/src/update/install-detection.mjs +73 -0
- package/src/update/job.ts +101 -49
- package/src/update/pnpm-global-install.d.mts +144 -0
- package/src/update/pnpm-global-install.mjs +591 -0
- package/src/update/pnpm-invocation.d.mts +43 -0
- package/src/update/pnpm-invocation.mjs +141 -0
- package/src/update/registry-integrity.d.mts +16 -0
- package/src/update/registry-integrity.mjs +37 -0
- package/src/update/transactional-install.d.mts +1 -1
- package/src/update/transactional-install.mjs +101 -7
- package/src/update/tray-update-plan.mjs +1 -1
- package/src/vision/plan.ts +13 -3
- package/src/vision/routed-describe.ts +51 -20
- package/src/web-search/ollama-executor.ts +127 -0
- package/src/web-search/passthrough-bridge.ts +761 -0
- package/gui/dist/assets/index-BtyONQrZ.js +0 -115
package/src/server/index.ts
CHANGED
|
@@ -17,6 +17,7 @@ import {
|
|
|
17
17
|
loadConfig,
|
|
18
18
|
saveConfig,
|
|
19
19
|
getConfigDir,
|
|
20
|
+
loopbackCompanionBindError,
|
|
20
21
|
websocketsEnabled,
|
|
21
22
|
} from "../config";
|
|
22
23
|
import { grokDefaultReasoningEffort } from "../grok/effort";
|
|
@@ -28,6 +29,7 @@ import { withCatalogWriteSerialization } from "../codex/catalog-write-serializat
|
|
|
28
29
|
import { invalidateCodexModelsCacheWithPermit } from "../codex/catalog/sync";
|
|
29
30
|
import { currentServiceHomes, serviceStatePathsForOpenCodexHome } from "../service";
|
|
30
31
|
import { shouldSyncCodexOnStart } from "../codex/desired-state";
|
|
32
|
+
import { effectiveLoopbackListenerPort } from "../codex/loopback-target";
|
|
31
33
|
import {
|
|
32
34
|
createWindowsTaskListingCache,
|
|
33
35
|
inspectNativeCodexOwnership,
|
|
@@ -63,7 +65,11 @@ import { runModelRenameStartupMigration } from "../providers/model-rename-startu
|
|
|
63
65
|
import { isCanonicalOpenAiForwardProvider, OPENAI_CODEX_PROVIDER_ID } from "../providers/openai-tiers";
|
|
64
66
|
import { providerCodexAccountMode } from "../providers/registry";
|
|
65
67
|
import type { StorageCleanupPolicy } from "../types";
|
|
66
|
-
import {
|
|
68
|
+
import {
|
|
69
|
+
MAX_CONFIGURABLE_INBOUND_BODY_BYTES,
|
|
70
|
+
MIN_CONFIGURABLE_INBOUND_BODY_BYTES,
|
|
71
|
+
resolveInboundBodyLimitBytes,
|
|
72
|
+
} from "./request-decompress";
|
|
67
73
|
import {
|
|
68
74
|
CodexAccountCooldownError,
|
|
69
75
|
cooldownErrorMessage,
|
|
@@ -769,8 +775,16 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
|
|
|
769
775
|
const bindHost = !configuredHost || /^localhost$/i.test(configuredHost) ? "127.0.0.1" : configuredHost;
|
|
770
776
|
|
|
771
777
|
// Unauthenticated loopback listener (#1102). Off unless explicitly enabled.
|
|
778
|
+
// A port-less enabled entry is the companion form: same port as the public listener, on
|
|
779
|
+
// 127.0.0.1 (#4236). Refuse an impossible pair here, before any bind, so a hand edit that
|
|
780
|
+
// bypassed validateConfigCandidate reports the collision rather than EADDRINUSE from a
|
|
781
|
+
// rollback that looks like a foreign process holding the port.
|
|
772
782
|
const loopbackListener = config.unauthenticatedLoopbackListener;
|
|
773
|
-
|
|
783
|
+
if (loopbackListener?.enabled === true && loopbackListener.port === undefined) {
|
|
784
|
+
const companionError = loopbackCompanionBindError(config.hostname, listenPort);
|
|
785
|
+
if (companionError) throw new Error(companionError);
|
|
786
|
+
}
|
|
787
|
+
const loopbackListenerPort = effectiveLoopbackListenerPort(config, listenPort);
|
|
774
788
|
// Hub management ingress is a third, management-only listener. Its address is intentionally
|
|
775
789
|
// fixed: the kernel loopback bind is the trust boundary that permits Tailscale identity headers.
|
|
776
790
|
const managementIngress = config.runtimeRole === "hub" ? config.hub?.managementIngress : undefined;
|
|
@@ -808,6 +822,25 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
|
|
|
808
822
|
* keeps the paid upstream behind its own admission and forward-credential checks, so admit only
|
|
809
823
|
* the exact methods and paths it serves (#3428).
|
|
810
824
|
*
|
|
825
|
+
* `POST /v1/messages` (Anthropic wire) and `POST /v1/chat/completions` (OpenAI chat wire)
|
|
826
|
+
* are the inference endpoints the hub's OWN local clients speak: `ocx claude` and the
|
|
827
|
+
* `system-env` injection and Claude Desktop 3P dial the first, Cursor Private Inference, the
|
|
828
|
+
* vision `routed-describe` helper and aside/opencode the second (#4236). On a hub whose
|
|
829
|
+
* public listener binds a tailnet address there is no other local socket for them, so
|
|
830
|
+
* leaving them off this list left every non-Codex local client pointed at a closed port.
|
|
831
|
+
* Both handlers resolve their own admission from the RECEIVING listener's policy view — the
|
|
832
|
+
* same resolver and the same loopback short-circuit `/v1/responses` already uses — so this
|
|
833
|
+
* adds a wire, not a trust level. `/api/*` is deliberately still absent: local management
|
|
834
|
+
* discovery goes to the authenticated management surface, never to this listener.
|
|
835
|
+
*
|
|
836
|
+
* `POST /v1/messages/count_tokens` completes that Anthropic wire. It is admitted on a
|
|
837
|
+
* narrower argument than the other two rather than on symmetry: it spends no provider quota,
|
|
838
|
+
* reaches no stored credential, and returns a token count computed from the request body the
|
|
839
|
+
* caller already holds. Withholding it bought no confinement — the same caller may POST the
|
|
840
|
+
* whole conversation to `/v1/messages` on this socket — and cost Claude Code its server-side
|
|
841
|
+
* count, which it then silently replaces with a local estimate. `/api/*`, `/healthz`,
|
|
842
|
+
* `/readyz` and the GUI remain 404 here, which is the boundary that actually matters.
|
|
843
|
+
*
|
|
811
844
|
* `GET /v1/models` is on the list for a reason that is easy to miss. When catalog
|
|
812
845
|
* materialization fails or finds no source, `syncCodex` warns and injects with
|
|
813
846
|
* `catalogPath: null`; Codex then builds an ONLINE model manager and `model/list` refreshes
|
|
@@ -820,6 +853,8 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
|
|
|
820
853
|
return req.method === "POST" || req.headers.get("upgrade")?.toLowerCase() === "websocket";
|
|
821
854
|
}
|
|
822
855
|
if (path === "/v1/responses/compact") return req.method === "POST";
|
|
856
|
+
if (path === "/v1/messages" || path === "/v1/chat/completions") return req.method === "POST";
|
|
857
|
+
if (path === "/v1/messages/count_tokens") return req.method === "POST";
|
|
823
858
|
if (path === "/v1/alpha/search") return req.method === "POST";
|
|
824
859
|
if (path === "/v1/images/generations" || path === "/v1/images/edits") {
|
|
825
860
|
return req.method === "POST";
|
|
@@ -1023,6 +1058,22 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
|
|
|
1023
1058
|
let loopbackServer: Server<WsData> | null = null;
|
|
1024
1059
|
let managementIngressServer: Server<WsData> | null = null;
|
|
1025
1060
|
|
|
1061
|
+
// Resolved once, before any listener binds. The clamp is silent inside the resolver so it
|
|
1062
|
+
// stays pure and per-request cheap; the operator is told here instead, once, because a
|
|
1063
|
+
// config value that was quietly reduced is exactly the thing they would otherwise debug
|
|
1064
|
+
// against the wrong limit.
|
|
1065
|
+
const inboundBodyLimitBytes = resolveInboundBodyLimitBytes(config.maxInboundBodyBytes);
|
|
1066
|
+
const requestedInboundBodyLimit = config.maxInboundBodyBytes;
|
|
1067
|
+
if (requestedInboundBodyLimit !== undefined
|
|
1068
|
+
&& requestedInboundBodyLimit > 0
|
|
1069
|
+
&& requestedInboundBodyLimit !== inboundBodyLimitBytes) {
|
|
1070
|
+
console.warn(
|
|
1071
|
+
`[server] maxInboundBodyBytes=${requestedInboundBodyLimit} is outside the supported range `
|
|
1072
|
+
+ `[${MIN_CONFIGURABLE_INBOUND_BODY_BYTES}, ${MAX_CONFIGURABLE_INBOUND_BODY_BYTES}]; `
|
|
1073
|
+
+ `using ${inboundBodyLimitBytes} bytes.`,
|
|
1074
|
+
);
|
|
1075
|
+
}
|
|
1076
|
+
|
|
1026
1077
|
type ServerIngress = "public" | "unauthenticated-loopback" | "hub-management";
|
|
1027
1078
|
function ingressForServer(requestServer: Server<WsData>): ServerIngress {
|
|
1028
1079
|
if (requestServer === loopbackServer) return "unauthenticated-loopback";
|
|
@@ -1042,7 +1093,10 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
|
|
|
1042
1093
|
userCostOverlayReconciler = startUserCostOverlayReconciler({ liveConfig: config });
|
|
1043
1094
|
const serveOptions = {
|
|
1044
1095
|
idleTimeout: 255,
|
|
1045
|
-
|
|
1096
|
+
// Bun rejects an oversized body before `fetch` runs, so the listener has to be raised
|
|
1097
|
+
// with the admission limit or the opt-in would do nothing. Fixed at bind time: a live
|
|
1098
|
+
// `maxInboundBodyBytes` edit needs a restart, which the config doc states.
|
|
1099
|
+
maxRequestBodySize: inboundBodyLimitBytes,
|
|
1046
1100
|
async fetch(req: Request, requestServer: Server<WsData>): Promise<Response> {
|
|
1047
1101
|
const ingress = ingressForServer(requestServer);
|
|
1048
1102
|
// The unauthenticated loopback listener (#1102) serves a fixed allowlist and nothing
|
|
@@ -1347,6 +1401,83 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
|
|
|
1347
1401
|
);
|
|
1348
1402
|
}
|
|
1349
1403
|
|
|
1404
|
+
if (url.pathname === "/v1/hub-state" && (req.method === "GET" || req.method === "HEAD")) {
|
|
1405
|
+
// #4236: a connected client had no way to learn which providers this hub can actually
|
|
1406
|
+
// serve, so `ocx status` on the client reported the CLIENT's empty credential store as
|
|
1407
|
+
// if it were the truth — "xai ✗ not logged in" on a machine whose hub has xAI logged
|
|
1408
|
+
// in. The fix is one least-privilege data-plane read, in the /v1/catalog (#809)
|
|
1409
|
+
// tradition: same admission resolver, same origin check, no parameters, no caller
|
|
1410
|
+
// credential forwarded upstream, and a body of booleans plus model ids. Widening
|
|
1411
|
+
// `/api/*` or handing the client an admin token to read `GET /api/providers` would
|
|
1412
|
+
// have traded a reporting defect for a credential one.
|
|
1413
|
+
//
|
|
1414
|
+
// What it discloses beyond /v1/catalog and /v1/models, exactly: `hasCredential`,
|
|
1415
|
+
// `loggedIn`, `authMode`, the featured roster, and the NAME and adapter of an ENABLED
|
|
1416
|
+
// provider those routes omit for want of a usable credential — which is the point of
|
|
1417
|
+
// the route. A `disabled` provider is NOT exported (`buildHubState` drops it), because
|
|
1418
|
+
// the catalog filters it out too and naming it here would be the only place a data key
|
|
1419
|
+
// learns of it.
|
|
1420
|
+
//
|
|
1421
|
+
// Placed between /v1/catalog and /v1/models so all three least-privilege client reads
|
|
1422
|
+
// stay in sight of each other.
|
|
1423
|
+
const admission = resolveApiAuth(req, policy);
|
|
1424
|
+
if (!admission) return withCors(formatErrorResponse(401, "authentication_error", "opencodex API key required"), req, policy);
|
|
1425
|
+
if (!isAllowedRequestOrigin(req, policy)) {
|
|
1426
|
+
return withCors(formatErrorResponse(403, "origin_rejected", "cross-origin data-plane request blocked"), req, policy);
|
|
1427
|
+
}
|
|
1428
|
+
// Role gate AFTER admission, deliberately: answering an unauthenticated caller would
|
|
1429
|
+
// turn this into a free "is that machine a hub?" probe. A standalone or client install
|
|
1430
|
+
// gains no surface at all — the route simply does not exist there.
|
|
1431
|
+
//
|
|
1432
|
+
// Built, not formatErrorResponse'd, for the same reason /v1/catalog builds its 404: the
|
|
1433
|
+
// code has to distinguish "this route exists and this host is not a hub" from "this
|
|
1434
|
+
// build has no such route", which is the difference between admission proof and a
|
|
1435
|
+
// vacuous pass in tests/server/api-key-attribution.test.ts.
|
|
1436
|
+
if (config.runtimeRole !== "hub") {
|
|
1437
|
+
return withCors(
|
|
1438
|
+
new Response(JSON.stringify({
|
|
1439
|
+
error: {
|
|
1440
|
+
type: "invalid_request_error",
|
|
1441
|
+
code: "hub_state_not_a_hub",
|
|
1442
|
+
message: "hub state is served only by a host whose runtimeRole is hub",
|
|
1443
|
+
},
|
|
1444
|
+
}), { status: 404, headers: { "content-type": "application/json" } }),
|
|
1445
|
+
req,
|
|
1446
|
+
policy,
|
|
1447
|
+
);
|
|
1448
|
+
}
|
|
1449
|
+
const { buildHubState } = await import("./hub-state");
|
|
1450
|
+
const { MAX_HUB_STATE_BYTES } = await import("../remote/hub-state");
|
|
1451
|
+
const { oauthLoginSummary } = await import("../oauth");
|
|
1452
|
+
// `true` masks emails, but the projection drops the field entirely; passing the mask
|
|
1453
|
+
// anyway means a future refactor that starts copying fields cannot leak a raw address.
|
|
1454
|
+
const body = JSON.stringify(buildHubState(config, oauthLoginSummary(true), VERSION));
|
|
1455
|
+
const bytes = Buffer.byteLength(body);
|
|
1456
|
+
if (bytes > MAX_HUB_STATE_BYTES) {
|
|
1457
|
+
return withCors(
|
|
1458
|
+
new Response(JSON.stringify({
|
|
1459
|
+
error: { type: "server_error", code: "hub_state_too_large", message: "hub state exceeds the maximum served size" },
|
|
1460
|
+
}), { status: 507, headers: { "content-type": "application/json" } }),
|
|
1461
|
+
req,
|
|
1462
|
+
policy,
|
|
1463
|
+
);
|
|
1464
|
+
}
|
|
1465
|
+
return withCors(
|
|
1466
|
+
new Response(req.method === "HEAD" ? null : body, {
|
|
1467
|
+
status: 200,
|
|
1468
|
+
headers: {
|
|
1469
|
+
"content-type": "application/json",
|
|
1470
|
+
// Varies by credential-bearing identity and by live login state: never cached,
|
|
1471
|
+
// and no validator to revalidate with (same rule as /v1/catalog).
|
|
1472
|
+
"cache-control": "no-store",
|
|
1473
|
+
"content-length": String(bytes),
|
|
1474
|
+
},
|
|
1475
|
+
}),
|
|
1476
|
+
req,
|
|
1477
|
+
policy,
|
|
1478
|
+
);
|
|
1479
|
+
}
|
|
1480
|
+
|
|
1350
1481
|
if (url.pathname === "/v1/models" && req.method === "GET") {
|
|
1351
1482
|
// #809: the catalog read sits immediately before model discovery because it shares
|
|
1352
1483
|
// that route's admission rationale exactly. Keep them adjacent so a future change to
|
|
@@ -1981,10 +2112,13 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
|
|
|
1981
2112
|
...admissionFields(admission),
|
|
1982
2113
|
inboundProtocol: "chat",
|
|
1983
2114
|
};
|
|
2115
|
+
// `policy`, not `config`: this route is now served on the unauthenticated loopback
|
|
2116
|
+
// listener too (#4236), and only the receiving listener's view produces CORS headers
|
|
2117
|
+
// that match the admission decision made above.
|
|
1984
2118
|
return runAdmittedHttpTurn(req, policy, async turnAdmissionLease => withCors(
|
|
1985
2119
|
await handleChatCompletions(req, config, logCtx, { requestId, start, turnAdmissionLease, admission }),
|
|
1986
2120
|
req,
|
|
1987
|
-
|
|
2121
|
+
policy,
|
|
1988
2122
|
));
|
|
1989
2123
|
}
|
|
1990
2124
|
|
|
@@ -2488,10 +2622,17 @@ export function startServer(port?: number, deps: StartServerDeps = {}): Server<W
|
|
|
2488
2622
|
// who forgot, has to be able to see that an unauthenticated surface is live without
|
|
2489
2623
|
// reading the file.
|
|
2490
2624
|
const loopbackPort = loopbackServer.port ?? loopbackListenerPort;
|
|
2491
|
-
|
|
2492
|
-
|
|
2493
|
-
|
|
2494
|
-
|
|
2625
|
+
if (loopbackListener?.enabled === true && loopbackListener.port === undefined) {
|
|
2626
|
+
// The companion form is the intended one-port hub topology, not a surprise surface: the
|
|
2627
|
+
// public listener is already on a non-loopback address, so this line states where local
|
|
2628
|
+
// processes go rather than warning about a second port nobody asked for.
|
|
2629
|
+
console.log(`🔁 Loopback companion active on http://127.0.0.1:${loopbackPort} — same port as the public listener; local processes need no credential`);
|
|
2630
|
+
} else {
|
|
2631
|
+
console.warn(`⚠️ Unauthenticated loopback listener active on http://127.0.0.1:${loopbackPort}`);
|
|
2632
|
+
console.warn(` Any local process can use it without a credential — it spends account`);
|
|
2633
|
+
console.warn(` quota and paid provider credentials, and can starve authenticated`);
|
|
2634
|
+
console.warn(` remote clients. Not for shared or multi-tenant hosts.`);
|
|
2635
|
+
}
|
|
2495
2636
|
}
|
|
2496
2637
|
|
|
2497
2638
|
if (managementIngressServer) {
|
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
import type { OcxConfig } from "../../types";
|
|
2
|
+
import { isWildcardHostname } from "../../codex/loopback-target";
|
|
3
|
+
import { localInferenceDestination } from "../../lib/local-destinations";
|
|
2
4
|
import { probeHostname } from "../proxy-liveness";
|
|
3
5
|
|
|
4
6
|
export interface ApiAccessEndpoints {
|
|
@@ -21,9 +23,14 @@ export type BuildApiAccessEndpointsOptions = {
|
|
|
21
23
|
requestOrigin?: string | null;
|
|
22
24
|
};
|
|
23
25
|
|
|
26
|
+
/**
|
|
27
|
+
* Wildcard bind scope, shared with `probeHostname` and the loopback-companion gate rather than
|
|
28
|
+
* re-spelled here: a third list of three spellings is how `0.0.0.0.` and `::0` ended up treated
|
|
29
|
+
* as specific bind addresses on one side and wildcards on the other.
|
|
30
|
+
*/
|
|
24
31
|
function isWildcardBindHost(hostname: string | undefined): boolean {
|
|
25
32
|
const trimmed = (hostname ?? "").trim();
|
|
26
|
-
return !trimmed || trimmed
|
|
33
|
+
return !trimmed || isWildcardHostname(trimmed);
|
|
27
34
|
}
|
|
28
35
|
|
|
29
36
|
/** Bracket bare IPv6 literals for URL authority composition. */
|
|
@@ -66,7 +73,7 @@ function originBaseUrl(raw: string): string | null {
|
|
|
66
73
|
* Falls back to loopback only when no usable request context is available.
|
|
67
74
|
*/
|
|
68
75
|
export function resolveApiAccessBaseUrl(
|
|
69
|
-
config: Pick<OcxConfig, "hostname" | "port">,
|
|
76
|
+
config: Pick<OcxConfig, "hostname" | "port" | "unauthenticatedLoopbackListener">,
|
|
70
77
|
opts: BuildApiAccessEndpointsOptions = {},
|
|
71
78
|
): string {
|
|
72
79
|
const port = config.port ?? 10100;
|
|
@@ -104,7 +111,11 @@ export function resolveApiAccessBaseUrl(
|
|
|
104
111
|
}
|
|
105
112
|
}
|
|
106
113
|
|
|
107
|
-
|
|
114
|
+
// Last resort: a wildcard bind with no usable request context, so the only address we can
|
|
115
|
+
// name is loopback — and on that address the unauthenticated loopback listener, when one is
|
|
116
|
+
// enabled, is the port a local caller should use (#4236). The branches above are unchanged:
|
|
117
|
+
// a specific bind or a real request host still describes the address the CLIENT reached.
|
|
118
|
+
return `${localInferenceDestination(config, port).origin}/v1`;
|
|
108
119
|
}
|
|
109
120
|
|
|
110
121
|
/** @deprecated Prefer resolveApiAccessBaseUrl; retained for focused host-format tests. */
|
|
@@ -110,7 +110,7 @@ import { applySystemEnvToggle } from "../system-env";
|
|
|
110
110
|
import { getCachedStartupHealth, invalidateStartupHealthCache } from "../startup-health-cache";
|
|
111
111
|
import { runWindowsTrayAction } from "../windows-tray-control";
|
|
112
112
|
import { runStartupInstallAction, type StartupInstallAction } from "../startup-action-control";
|
|
113
|
-
import { displayCodexRuntimePath, effortClampAppliesToRuntime, loadLastEffortClamp, resolveCodexRuntime } from "../../codex/runtime";
|
|
113
|
+
import { displayCodexRuntimePath, effortClampAppliesToRuntime, liveRemovedEfforts, loadLastEffortClamp, resolveCodexRuntime } from "../../codex/runtime";
|
|
114
114
|
|
|
115
115
|
import { isPlainRecord, parseDebugLogQuery, tokPerSecondResult, unavailableCostReason, costResult, requestLogDto, stripRegistryOnlyStaticHeaders, fetchAllModels } from "./shared";
|
|
116
116
|
import type { MetricUnavailableReason, TokPerSecondResult, CostEstimateReason, CostResult, MetricSource } from "./shared";
|
|
@@ -344,7 +344,7 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise<Respon
|
|
|
344
344
|
: null,
|
|
345
345
|
catalogClamp: {
|
|
346
346
|
active: clampActive,
|
|
347
|
-
removedEfforts: clampActive ? (lastClamp
|
|
347
|
+
removedEfforts: clampActive ? [...liveRemovedEfforts(lastClamp)] : [],
|
|
348
348
|
runtimeVersion: clampActive ? (lastClamp?.runtimeVersion ?? null) : null,
|
|
349
349
|
},
|
|
350
350
|
warning: warningParts.length > 0 ? warningParts.join(" ") : null,
|
|
@@ -14,6 +14,7 @@ import { cursorLastSeen, type CursorSeen } from "../../integrations/cursor-seen"
|
|
|
14
14
|
import { detectCursorInstalls, type CursorInstall } from "../../integrations/cursor-detect";
|
|
15
15
|
import { loadCursorEffortTable } from "../../integrations/cursor-effort-table";
|
|
16
16
|
import { configuredApiAuthToken, isApiAuthRequired, jsonResponse } from "../auth-cors";
|
|
17
|
+
import { localInferenceDestination } from "../../lib/local-destinations";
|
|
17
18
|
import { fetchAllModels } from "../management-api";
|
|
18
19
|
import { predictCursorEffort } from "../models-capabilities";
|
|
19
20
|
import { expandCursorEffortRow, knownEffortRowIds } from "../effort-row";
|
|
@@ -54,11 +55,19 @@ export async function buildCursorIntegrationStatus(
|
|
|
54
55
|
// The port the browser reached is the one Cursor on the same machine will reach too; the
|
|
55
56
|
// runtime record and config.port are fallbacks for a request that carries no port.
|
|
56
57
|
const port = runtime?.port ?? (Number(ctx.url?.port) || config.port);
|
|
57
|
-
//
|
|
58
|
-
//
|
|
58
|
+
// Cursor runs on this machine, so the gateway URL it is told to paste is the LOCAL one: the
|
|
59
|
+
// unauthenticated loopback listener when one is enabled, and otherwise the bind address on the
|
|
60
|
+
// public port — 127.0.0.1 for a loopback or wildcard bind exactly as before, and the tailnet
|
|
61
|
+
// or LAN address on a hub, where no loopback socket exists to paste (#4236).
|
|
62
|
+
const gateway = localInferenceDestination(config, port ?? 10100);
|
|
63
|
+
// apiKeyMode describes the admission rule of the destination just resolved, which on the
|
|
64
|
+
// loopback listener is "no key needed" and on every other form is "a key is required".
|
|
65
|
+
// Pasting one into the listener is harmless; omitting one on a bind that demands it is not.
|
|
59
66
|
const credentialConfigured = !!configuredApiAuthToken(config)
|
|
60
67
|
|| (config.apiKeys ?? []).some(entry => !!entry.key.trim());
|
|
61
|
-
const apiKeyMode = isApiAuthRequired(config) || credentialConfigured
|
|
68
|
+
const apiKeyMode = gateway.requiresAdmissionToken || isApiAuthRequired(config) || credentialConfigured
|
|
69
|
+
? "credential"
|
|
70
|
+
: "placeholder";
|
|
62
71
|
|
|
63
72
|
const limits = nativeContextLimits(config);
|
|
64
73
|
// Same visibility rules as the raw /v1/models list Cursor will read: disabled models and
|
|
@@ -107,7 +116,7 @@ export async function buildCursorIntegrationStatus(
|
|
|
107
116
|
},
|
|
108
117
|
regularCursor: { installed: regular !== undefined, path: regular?.path ?? null },
|
|
109
118
|
gateway: {
|
|
110
|
-
baseUrl:
|
|
119
|
+
baseUrl: `${gateway.origin}/v1`,
|
|
111
120
|
apiKeyMode,
|
|
112
121
|
placeholder: CURSOR_GATEWAY_PLACEHOLDER_KEY,
|
|
113
122
|
},
|
|
@@ -115,7 +115,10 @@ export async function handleLogsUsageRoutes(ctx: ManagementContext): Promise<Res
|
|
|
115
115
|
}
|
|
116
116
|
const all = getRequestLogEntries();
|
|
117
117
|
const total = filteredRequestLogCount(all, url.searchParams);
|
|
118
|
-
|
|
118
|
+
// Not point-free: requestLogDto takes an options object second, and Array.map would pass the
|
|
119
|
+
// element INDEX into it. An explicit arrow keeps the default (decode rate included) and is
|
|
120
|
+
// what /api/logs wants; /api/request-history opts out at its own call sites.
|
|
121
|
+
const logs = filterRequestLogs(all, url.searchParams).map(entry => requestLogDto(entry));
|
|
119
122
|
const poll = selectRequestLogPoll(logs, url.searchParams, cursor);
|
|
120
123
|
return jsonResponse({
|
|
121
124
|
timeZone: Intl.DateTimeFormat().resolvedOptions().timeZone,
|
|
@@ -147,10 +147,25 @@ export async function listManagementModelRows(
|
|
|
147
147
|
};
|
|
148
148
|
});
|
|
149
149
|
const publicModels = uniqueCatalogModelsForPublicList(models);
|
|
150
|
+
// Custom rows below are REBUILT from config.customModels rather than spread from a
|
|
151
|
+
// CatalogModel, so every field gather computed for the same slug has to be carried across by
|
|
152
|
+
// hand. Without this a custom model whose provider is out of credit would be the one row on
|
|
153
|
+
// the page that never shows as inactive (#1711), because the gather-derived row it replaces
|
|
154
|
+
// is dropped by the slug dedup below.
|
|
155
|
+
const quotaInactiveByNamespaced = new Map(
|
|
156
|
+
publicModels
|
|
157
|
+
.filter(model => model.quotaInactiveReason !== undefined)
|
|
158
|
+
.map(model => [catalogModelSlug(model), model.quotaInactiveReason!] as const),
|
|
159
|
+
);
|
|
150
160
|
const comboNamespaced = new Set(
|
|
151
161
|
publicModels.filter(model => model.provider === "combo").map(catalogModelSlug),
|
|
152
162
|
);
|
|
153
|
-
const visibleCustomModels = customModels
|
|
163
|
+
const visibleCustomModels = customModels
|
|
164
|
+
.filter(model => !comboNamespaced.has(model.namespaced))
|
|
165
|
+
.map(model => {
|
|
166
|
+
const quotaInactiveReason = quotaInactiveByNamespaced.get(model.namespaced);
|
|
167
|
+
return quotaInactiveReason ? { ...model, quotaInactiveReason } : model;
|
|
168
|
+
});
|
|
154
169
|
// Custom metadata wins when a physical live/static row resolves to the same Codex-facing
|
|
155
170
|
// slug, while a combo keeps the same precedence it has in routing and /v1/models.
|
|
156
171
|
const customNamespaced = new Set(visibleCustomModels.map(c => c.namespaced));
|
|
@@ -25,6 +25,7 @@ import {
|
|
|
25
25
|
} from "../../oauth";
|
|
26
26
|
import { OAuthMutationBusyError, removeCredential } from "../../oauth/store";
|
|
27
27
|
import { providerDestinationResolvedError } from "../../lib/destination-policy";
|
|
28
|
+
import { emailMaskingEnabled } from "../../lib/privacy";
|
|
28
29
|
import { reconcileLiveStateStores } from "../../lib/state-store-registrations";
|
|
29
30
|
import { enrichProviderFromCatalog, listKeyLoginProviders } from "../../oauth/key-providers";
|
|
30
31
|
import { deriveProviderPresets } from "../../providers/derive";
|
|
@@ -233,7 +234,10 @@ export async function handleOauthAccountRoutes(ctx: ManagementContext): Promise<
|
|
|
233
234
|
if (url.pathname === "/api/oauth/status" && req.method === "GET") {
|
|
234
235
|
const provider = (url.searchParams.get("provider") ?? "").trim().toLowerCase();
|
|
235
236
|
if (!isPublicOAuthProvider(provider)) return jsonResponse({ error: "unknown oauth provider" }, 400);
|
|
236
|
-
|
|
237
|
+
// Resolved here, at the request boundary that already holds the config, and passed down.
|
|
238
|
+
// getLoginStatus stays free of config I/O. This route does not re-mask afterwards: it
|
|
239
|
+
// consumes the already-projected status rather than redacting a second time.
|
|
240
|
+
const status = getLoginStatus(provider, emailMaskingEnabled(config));
|
|
237
241
|
return jsonResponse(status);
|
|
238
242
|
}
|
|
239
243
|
|
|
@@ -269,7 +273,7 @@ export async function handleOauthAccountRoutes(ctx: ManagementContext): Promise<
|
|
|
269
273
|
} = await import("../../oauth/health");
|
|
270
274
|
const projectAccounts = () => {
|
|
271
275
|
const set = getAccountSet(provider);
|
|
272
|
-
const current = getLoginStatus(provider);
|
|
276
|
+
const current = getLoginStatus(provider, emailMaskingEnabled(config));
|
|
273
277
|
return {
|
|
274
278
|
activeAccountId: current.activeAccountId ?? null,
|
|
275
279
|
accounts: (current.accounts ?? []).map(summary => {
|
|
@@ -54,6 +54,7 @@ import {
|
|
|
54
54
|
readBoundedDiscoveryJson,
|
|
55
55
|
resolveProviderModelDiscovery,
|
|
56
56
|
} from "../../providers/model-discovery";
|
|
57
|
+
import { extractGoogleAiStudioModelItems } from "../../providers/google-ai-studio-model-discovery";
|
|
57
58
|
import { routedSlug, slugEquals } from "../../providers/slug-codec";
|
|
58
59
|
import { clearAccountQuotaCache, clearProviderQuotaCache, fetchProviderQuotaReports } from "../../providers/quota";
|
|
59
60
|
import { clearKeyCooldowns } from "../../providers/key-failover";
|
|
@@ -1428,13 +1429,19 @@ export async function handleProviderRoutes(ctx: ManagementContext): Promise<Resp
|
|
|
1428
1429
|
return jsonResponse({ ok: false, latencyMs, error: "upstream CCA model discovery returned an unexpected shape" });
|
|
1429
1430
|
}
|
|
1430
1431
|
// OpenAI-style lists (and Together top-level arrays) use the same validation/dedupe/filter
|
|
1431
|
-
// as catalog discovery. Google
|
|
1432
|
-
//
|
|
1432
|
+
// as catalog discovery. Google AI Studio parses the native `models[]` envelope and filters to
|
|
1433
|
+
// `generateContent`, while other providers fall back to generic envelope rows if they return `models[]`.
|
|
1433
1434
|
const record = bounded.value !== null && typeof bounded.value === "object" && !Array.isArray(bounded.value)
|
|
1434
1435
|
? bounded.value as Record<string, unknown>
|
|
1435
1436
|
: undefined;
|
|
1437
|
+
const isAiStudio = effectiveGoogleMode(name, prov) === "ai-studio";
|
|
1438
|
+
const googleAiStudio = !ccaModels && isAiStudio
|
|
1439
|
+
? extractGoogleAiStudioModelItems(bounded.value, discovery.maxModels)
|
|
1440
|
+
: undefined;
|
|
1436
1441
|
const extracted = ccaModels
|
|
1437
1442
|
? undefined
|
|
1443
|
+
: googleAiStudio?.ok
|
|
1444
|
+
? googleAiStudio
|
|
1438
1445
|
: Array.isArray(bounded.value) || Array.isArray(record?.data)
|
|
1439
1446
|
? extractProviderModelItems(bounded.value, discovery)
|
|
1440
1447
|
: extractModelEnvelopeRows(bounded.value, discovery.maxModels, ["models"]);
|
|
@@ -106,7 +106,9 @@ export async function handleRequestHistoryRoutes(ctx: ManagementContext): Promis
|
|
|
106
106
|
to,
|
|
107
107
|
}, cursor, limit);
|
|
108
108
|
return jsonResponse({
|
|
109
|
-
|
|
109
|
+
// The decode rate is a Logs-page metric; this endpoint shares the DTO but not its
|
|
110
|
+
// contract, so it opts out rather than silently widening its own response shape (#4038).
|
|
111
|
+
entries: page.rows.map(row => requestLogDto(requestLogEntryFromPersistedUsage(row), { includeDecodeRate: false })),
|
|
110
112
|
...(page.nextCursor ? { nextCursor: page.nextCursor } : {}),
|
|
111
113
|
hasMore: page.hasMore,
|
|
112
114
|
index: {
|
|
@@ -184,7 +186,7 @@ export async function handleRequestHistoryRoutes(ctx: ManagementContext): Promis
|
|
|
184
186
|
if (!entry) {
|
|
185
187
|
return jsonResponse({ error: { code: "not_found", message: "unknown request" } }, 404, req, config);
|
|
186
188
|
}
|
|
187
|
-
return jsonResponse(requestLogDto(requestLogEntryFromPersistedUsage(entry)), 200, req, config);
|
|
189
|
+
return jsonResponse(requestLogDto(requestLogEntryFromPersistedUsage(entry), { includeDecodeRate: false }), 200, req, config);
|
|
188
190
|
}
|
|
189
191
|
|
|
190
192
|
return null;
|
|
@@ -95,6 +95,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
|
|
|
95
95
|
{ method: "PATCH", path: "/api/codex-auth/pool-strategy", module: "codex/auth-api", mutates: true },
|
|
96
96
|
{ method: "POST", path: "/api/codex-auth/accounts", module: "codex/auth-api", mutates: true },
|
|
97
97
|
{ method: "POST", path: "/api/codex-auth/accounts/clear-cooldown", module: "codex/auth-api", mutates: true },
|
|
98
|
+
{ method: "POST", path: "/api/codex-auth/accounts/refresh", module: "codex/auth-api", mutates: true },
|
|
98
99
|
{ method: "POST", path: "/api/codex-auth/login", module: "codex/auth-api", mutates: true },
|
|
99
100
|
{ method: "POST", path: "/api/codex-auth/login/cancel", module: "codex/auth-api", mutates: true },
|
|
100
101
|
{ method: "POST", path: "/api/codex-auth/login/code", module: "codex/auth-api", mutates: true },
|
|
@@ -148,9 +149,9 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
|
|
|
148
149
|
{ method: "GET", path: "/api/client-integrations/aside/profiles/{profileId}", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
|
|
149
150
|
{ method: "PUT", path: "/api/client-integrations/aside/profiles/{profileId}", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode" },
|
|
150
151
|
{ method: "GET", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
|
|
151
|
-
{ method: "DELETE", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside history deletion uses the dashboard journal cleanup; the CLI has history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/
|
|
152
|
+
{ method: "DELETE", path: "/api/client-integrations/aside/profiles/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside history deletion uses the dashboard journal cleanup; the CLI has history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
|
|
152
153
|
{ method: "GET", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: false, mechanism: "prefix-decode" },
|
|
153
|
-
{ method: "DELETE", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside profile history deletion uses the dashboard journal cleanup; the CLI has scoped history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/
|
|
154
|
+
{ method: "DELETE", path: "/api/client-integrations/aside/profiles/{profileId}/journal", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode", exempt: { reason: "deferred-verb", why: "Aside profile history deletion uses the dashboard journal cleanup; the CLI has scoped history and restore but no deletion verb yet.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
|
|
154
155
|
{ method: "POST", path: "/api/client-integrations/aside/profiles/{profileId}/restore", module: "server/management/aside-profile-routes", mutates: true, mechanism: "prefix-decode" },
|
|
155
156
|
// server/management/codex-prompt-routes
|
|
156
157
|
{ method: "GET", path: "/api/codex-prompt", module: "server/management/codex-prompt-routes", mutates: false },
|
|
@@ -186,7 +187,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
|
|
|
186
187
|
// server/management/integration-routes
|
|
187
188
|
{ method: "GET", path: "/api/client-integrations", module: "server/management/integration-routes", mutates: false },
|
|
188
189
|
{ method: "GET", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: false },
|
|
189
|
-
{ method: "DELETE", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Retiring one rollback row is a dashboard-local cleanup; the CLI verb that would drive it is owed by a later work-phase and is not implemented here.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/
|
|
190
|
+
{ method: "DELETE", path: "/api/client-integrations/journal", module: "server/management/integration-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Retiring one rollback row is a dashboard-local cleanup; the CLI verb that would drive it is owed by a later work-phase and is not implemented here.", owner: "260904_priority65_closeout WP7", ownerDoc: "devlog/_fin/260904_priority65_closeout/060_wp7_rollback_journal_crud.md" } },
|
|
190
191
|
{ method: "POST", path: "/api/client-integrations/restore", module: "server/management/integration-routes", mutates: true },
|
|
191
192
|
// server/management/lab-automation-routes
|
|
192
193
|
{ method: "GET", path: "/api/lab/automation", module: "server/management/lab-automation-routes", mutates: false, exempt: { reason: "local-transport", why: "ocx lab reads the same rows from the local SQLite projection; src/cli/lab.ts imports ../lab/query directly and never fetches /api/lab." } },
|
|
@@ -292,7 +293,7 @@ export const MANAGEMENT_ROUTES: readonly ManagementRoute[] = [
|
|
|
292
293
|
{ method: "PATCH", path: "/api/providers", module: "server/management/provider-routes", mutates: true },
|
|
293
294
|
{ method: "POST", path: "/api/providers", module: "server/management/provider-routes", mutates: true },
|
|
294
295
|
{ method: "POST", path: "/api/providers/test", module: "server/management/provider-routes", mutates: true },
|
|
295
|
-
{ method: "PUT", path: "/api/providers", module: "server/management/provider-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Issue #3280 scopes this atomic batch endpoint to the GUI JSON editor; a matching CLI verb is outside wp5 and remains owed.", owner: "wp5-followup", ownerDoc: "devlog/
|
|
296
|
+
{ method: "PUT", path: "/api/providers", module: "server/management/provider-routes", mutates: true, exempt: { reason: "deferred-verb", why: "Issue #3280 scopes this atomic batch endpoint to the GUI JSON editor; a matching CLI verb is outside wp5 and remains owed.", owner: "wp5-followup", ownerDoc: "devlog/_fin/260903_bug_drawdown_bcda/050_phase5.md" } },
|
|
296
297
|
{ method: "PUT", path: "/api/provider-context-caps", module: "server/management/provider-routes", mutates: true },
|
|
297
298
|
// server/management/quota-reset-routes
|
|
298
299
|
{ method: "GET", path: "/api/quota-resets", module: "server/management/quota-reset-routes", mutates: false, mechanism: "negated-guard" },
|
|
@@ -79,7 +79,8 @@ export function parseDebugLogQuery(url: URL): { after: number; limit: number } {
|
|
|
79
79
|
export type MetricUnavailableReason =
|
|
80
80
|
| "usage_missing" | "usage_unsupported" | "output_missing" | "invalid_duration"
|
|
81
81
|
| "price_unmatched" | "invalid_cache_breakdown"
|
|
82
|
-
| "invalid_usage" | "combo_attempt_unavailable"
|
|
82
|
+
| "invalid_usage" | "combo_attempt_unavailable"
|
|
83
|
+
| "ttft_missing" | "decode_window_too_short";
|
|
83
84
|
|
|
84
85
|
export type TokPerSecondResult =
|
|
85
86
|
| { kind: "value"; value: number; estimated: boolean }
|
|
@@ -96,7 +97,7 @@ export type CostResult =
|
|
|
96
97
|
| { kind: "value"; estimate: NonNullable<ReturnType<typeof estimateRequestCost>>; estimateReasons: CostEstimateReason[] }
|
|
97
98
|
| { kind: "unavailable"; reason: MetricUnavailableReason };
|
|
98
99
|
|
|
99
|
-
export type MetricSource = Pick<RequestLogEntry, "provider" | "model" | "durationMs" | "usageStatus" | "usage" | "requestedServiceTier" | "configuredServiceTier" | "responseServiceTier" | "tierOutcome" | "routeDecision"> & {
|
|
100
|
+
export type MetricSource = Pick<RequestLogEntry, "provider" | "model" | "durationMs" | "firstOutputMs" | "usageStatus" | "usage" | "requestedServiceTier" | "configuredServiceTier" | "responseServiceTier" | "tierOutcome" | "routeDecision"> & {
|
|
100
101
|
attempts?: readonly PersistedUsageAttempt[];
|
|
101
102
|
};
|
|
102
103
|
|
|
@@ -113,6 +114,51 @@ export function tokPerSecondResult(entry: Pick<MetricSource, "durationMs" | "usa
|
|
|
113
114
|
return { kind: "value", value, estimated: entry.usageStatus === "estimated" || entry.usage.estimated === true };
|
|
114
115
|
}
|
|
115
116
|
|
|
117
|
+
/**
|
|
118
|
+
* Shortest post-TTFT window that can carry a decode-rate estimate (#4038).
|
|
119
|
+
*
|
|
120
|
+
* Below one second the window is dominated by things that are not decoding: TTFT jitter, the
|
|
121
|
+
* proxy's own buffering, and the granularity of the timestamps themselves. A 240-token response
|
|
122
|
+
* whose first token arrived 50 ms before the last one is not a 4800 tok/s model, and printing
|
|
123
|
+
* that number is worse than printing nothing — which is precisely why the earlier attempt at
|
|
124
|
+
* this metric (#4040) was closed as an unreliable estimate.
|
|
125
|
+
*/
|
|
126
|
+
export const MIN_DECODE_WINDOW_MS = 1_000;
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* Estimated DECODE throughput: output tokens over the window after the first token (#4038).
|
|
130
|
+
*
|
|
131
|
+
* Strictly additive. `tokensPerSecond`, `tokPerSecondResult`, `RequestLogEntry` and
|
|
132
|
+
* `usage.jsonl` are untouched, and the end-to-end rate beside it keeps meaning exactly what it
|
|
133
|
+
* has always meant — it is documented as end-to-end, so this is a missing metric rather than a
|
|
134
|
+
* miscalculated one.
|
|
135
|
+
*
|
|
136
|
+
* Always `estimated: true`. The proxy's TTFT is when the FIRST BYTE reached the proxy, which is
|
|
137
|
+
* not the provider's own generation start, so this can never be more than an estimate no matter
|
|
138
|
+
* how long the window is. Saying so in the payload is the honest half of the answer to #4040;
|
|
139
|
+
* MIN_DECODE_WINDOW_MS is the other half.
|
|
140
|
+
*/
|
|
141
|
+
export function decodeTokPerSecondResult(
|
|
142
|
+
entry: Pick<MetricSource, "durationMs" | "firstOutputMs" | "usageStatus" | "usage">,
|
|
143
|
+
): TokPerSecondResult {
|
|
144
|
+
if (!entry.usage) return { kind: "unavailable", reason: "usage_missing" };
|
|
145
|
+
if (entry.usageStatus === "unsupported") return { kind: "unavailable", reason: "usage_unsupported" };
|
|
146
|
+
if (entry.usage.outputTokens <= 0) return { kind: "unavailable", reason: "output_missing" };
|
|
147
|
+
// A row that predates TTFT capture, or a non-streaming turn that never recorded one, has no
|
|
148
|
+
// window to measure. That is a different fact from a bad duration, so it gets its own reason.
|
|
149
|
+
if (entry.firstOutputMs === undefined) return { kind: "unavailable", reason: "ttft_missing" };
|
|
150
|
+
if (!Number.isFinite(entry.firstOutputMs) || entry.firstOutputMs < 0 || !Number.isFinite(entry.durationMs)) {
|
|
151
|
+
return { kind: "unavailable", reason: "invalid_duration" };
|
|
152
|
+
}
|
|
153
|
+
const windowMs = entry.durationMs - entry.firstOutputMs;
|
|
154
|
+
// TTFT at or past the total duration means the two clocks disagree; there is no window.
|
|
155
|
+
if (windowMs <= 0) return { kind: "unavailable", reason: "invalid_duration" };
|
|
156
|
+
if (windowMs < MIN_DECODE_WINDOW_MS) return { kind: "unavailable", reason: "decode_window_too_short" };
|
|
157
|
+
const value = tokensPerSecond(entry.usage.outputTokens, windowMs);
|
|
158
|
+
if (value === null) return { kind: "unavailable", reason: "invalid_duration" };
|
|
159
|
+
return { kind: "value", value, estimated: true };
|
|
160
|
+
}
|
|
161
|
+
|
|
116
162
|
export function unavailableCostReason(entry: MetricSource): MetricUnavailableReason {
|
|
117
163
|
// Normalizer-first classification: the landed normalizer recovers legacy
|
|
118
164
|
// cachedInputTokens=read+write rows via retry, so a raw read+write>input
|
|
@@ -152,11 +198,26 @@ export function costResult(entry: MetricSource): CostResult {
|
|
|
152
198
|
return { kind: "value", estimate, estimateReasons };
|
|
153
199
|
}
|
|
154
200
|
|
|
155
|
-
|
|
201
|
+
/**
|
|
202
|
+
* `/api/logs` row projection.
|
|
203
|
+
*
|
|
204
|
+
* `includeDecodeRate` exists because `/api/request-history` shares this DTO but not its
|
|
205
|
+
* contract (#4038). The value would be meaningful there — `firstOutputMs` does survive into a
|
|
206
|
+
* persisted-usage row — so this is a scope decision, not a correctness one: the decode rate is
|
|
207
|
+
* a Logs-page metric, and widening a separate endpoint's response shape is not this change's
|
|
208
|
+
* business. Flipping it on later is one argument.
|
|
209
|
+
*/
|
|
210
|
+
export function requestLogDto(
|
|
211
|
+
entry: RequestLogEntry,
|
|
212
|
+
{ includeDecodeRate = true }: { includeDecodeRate?: boolean } = {},
|
|
213
|
+
): Record<string, unknown> {
|
|
156
214
|
return {
|
|
157
215
|
...entry,
|
|
158
216
|
displayMetrics: {
|
|
159
217
|
tokPerSecond: tokPerSecondResult(entry),
|
|
218
|
+
// The parent uses the REQUEST's own TTFT. A combo parent must not borrow an attempt's,
|
|
219
|
+
// which would measure a window the parent never had.
|
|
220
|
+
...(includeDecodeRate ? { decodeTokPerSecond: decodeTokPerSecondResult(entry) } : {}),
|
|
160
221
|
cost: costResult(entry),
|
|
161
222
|
},
|
|
162
223
|
...(entry.attempts?.length
|
|
@@ -165,6 +226,8 @@ export function requestLogDto(entry: RequestLogEntry): Record<string, unknown> {
|
|
|
165
226
|
...attempt,
|
|
166
227
|
displayMetrics: {
|
|
167
228
|
tokPerSecond: tokPerSecondResult(attempt),
|
|
229
|
+
// Each attempt measures its own attempt-relative TTFT.
|
|
230
|
+
...(includeDecodeRate ? { decodeTokPerSecond: decodeTokPerSecondResult(attempt) } : {}),
|
|
168
231
|
cost: costResult({ ...attempt, attempts: undefined, routeDecision: entry.routeDecision, requestedServiceTier: entry.requestedServiceTier, configuredServiceTier: entry.configuredServiceTier, responseServiceTier: entry.responseServiceTier }),
|
|
169
232
|
},
|
|
170
233
|
})),
|
|
@@ -385,7 +385,7 @@ export async function handleManagementAPI(
|
|
|
385
385
|
const { ConfigMutationLockError } = await import("../config");
|
|
386
386
|
const { CodexCredentialRefreshLockTimeoutError } = await import("../codex/account-store");
|
|
387
387
|
try {
|
|
388
|
-
return await handleCodexAuthAPI(req, url, config, convergeCodexCatalog);
|
|
388
|
+
return await handleCodexAuthAPI(req, url, config, convergeCodexCatalog, principal);
|
|
389
389
|
} catch (error) {
|
|
390
390
|
// Credential writers remap ConfigMutationLockError to CodexCredentialRefreshLockTimeoutError;
|
|
391
391
|
// treat both as the same retryable busy response.
|