codex-grok-bridge 1.6.0 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,15 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
+ ## 1.7.0 — 2026-09-29
6
+
7
+ - One `x-grok-conv-id` per Codex thread, and the forwarded transcript prefix stays byte-stable so prompt cache can hit. The full transcript is still sent.
8
+ - Upstream `cached_prompt_tokens` and `cache_read_input_tokens` are copied onto `response.completed` usage and the diagnostics log when the proxy sends them, including `0`. Missing counters are not invented.
9
+
10
+ ## 1.6.1 — 2026-09-28
11
+
12
+ - Darwin bundled CLI resolves `Codex.app/Contents/Resources/codex-cli/bin/codex` when that file exists (Codex 26.924). The legacy `Contents/Resources/codex` path remains the fallback. `CODEX_BINARY` still wins. Linux and Windows layouts are unchanged.
13
+
5
14
  ## 1.6.0 — 2026-09-27
6
15
 
7
16
  - Default catalog model is `grok-4.7` (`Grok 4.7 / xAI`). `grok-4.6` stays listed so existing threads still resolve.
package/README.md CHANGED
@@ -130,8 +130,11 @@ older CLI envelope. That path pastes the whole JSON into a prompt each turn, so
130
130
  it is slower and more expensive, has no token-by-token streaming, and is capped
131
131
  at three minutes per request.
132
132
 
133
- The default Responses path streams. Codex `prompt_cache_key` is forwarded as
134
- `x-grok-conv-id`.
133
+ The default Responses path streams. One Codex thread keeps a single
134
+ `x-grok-conv-id`: the `thread-id` header when Codex sends it, otherwise
135
+ `prompt_cache_key`. That same id is written into `prompt_cache_key`. An
136
+ unchanged transcript prefix is resent byte-for-byte; the new items are
137
+ appended, not substituted for the history.
135
138
 
136
139
  ## What works and what does not
137
140
 
@@ -453,8 +456,10 @@ Responses 경로가 이상하면 `GROK_BRIDGE_INFERENCE=cli`로 이전 CLI 봉
453
456
  씁니다. 매 턴 전체 JSON을 프롬프트로 넣으므로 더 느리고 비싸며, 토큰 단위
454
457
  실시간 출력이 없고, 요청당 3분 제한입니다.
455
458
 
456
- 기본 Responses 경로는 스트림을 전달합니다. Codex `prompt_cache_key`는
457
- `x-grok-conv-id`로 넘깁니다.
459
+ 기본 Responses 경로는 스트림을 전달합니다. Codex 스레드마다 `x-grok-conv-id`는
460
+ 하나입니다. `thread-id` 헤더가 있으면 그 값이고, 없으면 `prompt_cache_key`입니다.
461
+ 같은 id를 `prompt_cache_key`에도 넣습니다. 바뀌지 않은 트랜스크립트 접두는
462
+ 바이트 그대로 다시 보내고, 새 항목은 그 뒤에 붙입니다.
458
463
 
459
464
  ## 확인된 동작과 제한
460
465
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "codex-grok-bridge",
3
- "version": "1.6.0",
3
+ "version": "1.7.0",
4
4
  "description": "Run Grok 4.7 as the model inside Codex, with Codex still owning tools, permissions, history and MCP. Uses the grok login session, not an API key.",
5
5
  "type": "module",
6
6
  "license": "MIT",
package/src/bridge.mjs CHANGED
@@ -9,6 +9,12 @@ import {
9
9
  runGrok,
10
10
  } from "./cli-inference.mjs";
11
11
  import { openProxyStreamWithRetry, pipeProxySse } from "./proxy.mjs";
12
+ import {
13
+ applyCacheUsage,
14
+ createPrefixMemory,
15
+ readCacheUsage,
16
+ stableConvId,
17
+ } from "./prefix.mjs";
12
18
  import { toProxyRequest } from "./tools.mjs";
13
19
  import { classifyBridgeError, errorSignature, bridgeErrorMessage } from "./errors.mjs";
14
20
  import { createDiagnostics } from "./diagnostics.mjs";
@@ -36,6 +42,7 @@ export function createBridgeServer(options = {}) {
36
42
  limit: options.maxConcurrentInference,
37
43
  queueLimit: options.maxQueuedInference,
38
44
  });
45
+ const prefixes = createPrefixMemory();
39
46
  return http.createServer(async (req, res) => {
40
47
  const json = (status, data) => {
41
48
  res.writeHead(status, { "content-type": "application/json" });
@@ -112,6 +119,7 @@ export function createBridgeServer(options = {}) {
112
119
  const queuedAt = Date.now();
113
120
  let acquired = false;
114
121
  let queuedMs = 0;
122
+ let cacheUsage = null;
115
123
  try {
116
124
  await slots.acquire(controller.signal);
117
125
  acquired = true;
@@ -135,14 +143,18 @@ export function createBridgeServer(options = {}) {
135
143
  parsed.usage?.input_tokens ?? parsed.usage?.inputTokens ?? 0,
136
144
  outputTokens =
137
145
  parsed.usage?.output_tokens ?? parsed.usage?.outputTokens ?? 0;
146
+ const usage = {
147
+ input_tokens: inputTokens,
148
+ output_tokens: outputTokens,
149
+ total_tokens: inputTokens + outputTokens,
150
+ };
151
+ cacheUsage = readCacheUsage(parsed.usage);
152
+ if (cacheUsage) Object.assign(usage, cacheUsage);
153
+ applyCacheUsage(usage);
138
154
  event("response.completed", {
139
155
  response: {
140
156
  id,
141
- usage: {
142
- input_tokens: inputTokens,
143
- output_tokens: outputTokens,
144
- total_tokens: inputTokens + outputTokens,
145
- },
157
+ usage,
146
158
  },
147
159
  });
148
160
  } else {
@@ -150,7 +162,15 @@ export function createBridgeServer(options = {}) {
150
162
  // Grok takes input_image blocks natively; only unusable attachments
151
163
  // are swapped for an explanation, so one bad image cannot make the
152
164
  // upstream reject the whole conversation.
153
- const { request, map } = toProxyRequest(sanitizeImages(body));
165
+ const threadId = req.headers["thread-id"];
166
+ const convId = stableConvId(
167
+ Array.isArray(threadId) ? threadId[0] : threadId,
168
+ body.prompt_cache_key,
169
+ );
170
+ const { request, map, projected } = toProxyRequest(sanitizeImages(body));
171
+ if (convId) request.prompt_cache_key = convId;
172
+ request.input = prefixes.reuse(convId, projected);
173
+ const usageBox = {};
154
174
  const proxy = await openProxyStreamWithRetry({
155
175
  token: session.token,
156
176
  userId: session.userId,
@@ -158,8 +178,8 @@ export function createBridgeServer(options = {}) {
158
178
  signal: controller.signal,
159
179
  fetchImpl: options.proxyFetch,
160
180
  baseUrl: options.proxyBaseUrl,
161
- convId: body.prompt_cache_key,
162
- sessionId: req.headers["thread-id"],
181
+ convId,
182
+ sessionId: Array.isArray(threadId) ? threadId[0] : threadId,
163
183
  onRetry: ({ attempt, kind }) =>
164
184
  diagnostics.record({
165
185
  event: "turn_retried",
@@ -168,11 +188,16 @@ export function createBridgeServer(options = {}) {
168
188
  elapsedMs: Date.now() - startedAt,
169
189
  }),
170
190
  });
171
- await pipeProxySse(proxy.body, res, map);
191
+ try {
192
+ await pipeProxySse(proxy.body, res, map, usageBox);
193
+ } finally {
194
+ cacheUsage = usageBox.cacheUsage ?? null;
195
+ }
172
196
  }
173
197
  diagnostics.record({
174
198
  event: "turn_ok",
175
199
  ...shape,
200
+ ...(cacheUsage ?? {}),
176
201
  queuedMs,
177
202
  elapsedMs: Date.now() - startedAt,
178
203
  });
@@ -183,6 +208,7 @@ export function createBridgeServer(options = {}) {
183
208
  kind,
184
209
  signature: errorSignature(error),
185
210
  ...shape,
211
+ ...(cacheUsage ?? {}),
186
212
  queuedMs,
187
213
  elapsedMs: Date.now() - startedAt,
188
214
  detail: error?.message,
package/src/paths.mjs CHANGED
@@ -4,6 +4,8 @@ import { homedir } from "node:os";
4
4
 
5
5
  export const LINUX_CODEX_BINARY = "/usr/lib/chatgpt/resources/codex";
6
6
  export const LINUX_CHATGPT_BIN = "/usr/lib/chatgpt/ChatGPT";
7
+ export const DARWIN_CODEX_CLI_BINARY =
8
+ "/Applications/Codex.app/Contents/Resources/codex-cli/bin/codex";
7
9
  export const DARWIN_CODEX_BINARY = "/Applications/Codex.app/Contents/Resources/codex";
8
10
  export const DARWIN_CODEX_APP = "/Applications/Codex.app";
9
11
  export const LINUX_STOCK_PREFIX = "/usr/lib/chatgpt";
@@ -54,6 +56,8 @@ export function resolveCodexBinary(options = {}) {
54
56
  if (platform === "win32") {
55
57
  return resolveWin32Binary("codex.exe", "store-codex.txt", { ...options, env });
56
58
  }
59
+ const exists = options.existsSync ?? existsSync;
60
+ if (exists(DARWIN_CODEX_CLI_BINARY)) return DARWIN_CODEX_CLI_BINARY;
57
61
  return DARWIN_CODEX_BINARY;
58
62
  }
59
63
 
package/src/prefix.mjs ADDED
@@ -0,0 +1,163 @@
1
+ import { createHash } from "node:crypto";
2
+
3
+ // xAI prompt cache hits only when later requests repeat the same prefix
4
+ // unchanged, on the same x-grok-conv-id. Codex resends the whole transcript,
5
+ // but a fresh rewrite (new ids, encrypted reasoning, tool index names) makes
6
+ // that prefix differ. This module pins the conv id and reuses the exact
7
+ // items already forwarded.
8
+
9
+ const CACHE_USAGE_FIELDS = [
10
+ "cached_prompt_tokens",
11
+ "cache_read_input_tokens",
12
+ "cache_creation_input_tokens",
13
+ ];
14
+
15
+ const VOLATILE_KEYS = new Set([
16
+ "id",
17
+ "status",
18
+ "encrypted_content",
19
+ "encrypted_function_args",
20
+ ]);
21
+
22
+ export function stableConvId(threadId, promptCacheKey) {
23
+ const thread = typeof threadId === "string" ? threadId.trim() : "";
24
+ if (thread) return thread;
25
+ const cache = typeof promptCacheKey === "string" ? promptCacheKey.trim() : "";
26
+ return cache || null;
27
+ }
28
+
29
+ export function stableProxyName(spec, sanitizedName) {
30
+ const basis = `${spec.kind}\0${spec.namespace ?? ""}\0${spec.name ?? ""}`;
31
+ const hash = createHash("sha256").update(basis).digest("hex").slice(0, 8);
32
+ return `codex_${hash}_${sanitizedName}`;
33
+ }
34
+
35
+ function normalizeForFingerprint(value) {
36
+ if (Array.isArray(value)) return value.map(normalizeForFingerprint);
37
+ if (!value || typeof value !== "object") return value;
38
+ if (
39
+ value.type === "input_text" ||
40
+ value.type === "output_text" ||
41
+ value.type === "summary_text"
42
+ ) {
43
+ return {
44
+ type: value.type,
45
+ text: typeof value.text === "string" ? value.text : "",
46
+ };
47
+ }
48
+ if (value.type === "input_image") {
49
+ const url =
50
+ value.image_url && typeof value.image_url === "object"
51
+ ? value.image_url.url
52
+ : value.image_url;
53
+ const image = { type: "input_image" };
54
+ if (url !== undefined) image.image_url = url;
55
+ if (value.detail !== undefined) image.detail = value.detail;
56
+ return image;
57
+ }
58
+ const out = {};
59
+ for (const key of Object.keys(value).sort()) {
60
+ if (VOLATILE_KEYS.has(key) || key.startsWith("internal_")) continue;
61
+ const child = value[key];
62
+ if (child === undefined || child === null) continue;
63
+ out[key] = normalizeForFingerprint(child);
64
+ }
65
+ if (
66
+ out.role &&
67
+ out.content !== undefined &&
68
+ (out.type == null || out.type === "message")
69
+ ) {
70
+ out.type = "message";
71
+ if (typeof out.content === "string")
72
+ out.content = [{ type: "input_text", text: out.content }];
73
+ }
74
+ return out;
75
+ }
76
+
77
+ export function logicalFingerprint(item) {
78
+ return JSON.stringify(normalizeForFingerprint(item));
79
+ }
80
+
81
+ // Object key order is part of the cached prefix. Codex does not promise it.
82
+ export function stableJsonValue(value) {
83
+ if (Array.isArray(value)) return value.map(stableJsonValue);
84
+ if (!value || typeof value !== "object") return value;
85
+ const out = {};
86
+ for (const key of Object.keys(value).sort())
87
+ out[key] = stableJsonValue(value[key]);
88
+ return out;
89
+ }
90
+
91
+ const MAX_THREADS = 64;
92
+
93
+ export function createPrefixMemory() {
94
+ const threads = new Map();
95
+ return {
96
+ reuse(convId, pairs) {
97
+ const fresh = pairs.map((pair) => pair.item);
98
+ if (!convId || pairs.length === 0) return fresh;
99
+ const previous = threads.get(convId);
100
+ let reused = 0;
101
+ if (previous) {
102
+ const limit = Math.min(previous.length, pairs.length);
103
+ while (
104
+ reused < limit &&
105
+ previous[reused].fingerprint === pairs[reused].fingerprint
106
+ )
107
+ reused += 1;
108
+ }
109
+ const stored =
110
+ reused > 0
111
+ ? [...previous.slice(0, reused), ...pairs.slice(reused)]
112
+ : pairs.slice();
113
+ if (threads.has(convId)) threads.delete(convId);
114
+ threads.set(
115
+ convId,
116
+ stored.map((pair) => ({
117
+ fingerprint: pair.fingerprint,
118
+ item: structuredClone(pair.item),
119
+ })),
120
+ );
121
+ while (threads.size > MAX_THREADS) {
122
+ const oldest = threads.keys().next().value;
123
+ threads.delete(oldest);
124
+ }
125
+ if (!previous || reused === 0) return fresh;
126
+ return [
127
+ ...previous.slice(0, reused).map((pair) => structuredClone(pair.item)),
128
+ ...pairs.slice(reused).map((pair) => pair.item),
129
+ ];
130
+ },
131
+ };
132
+ }
133
+
134
+ // Copy only cache counters the upstream actually sent. Zero is a real miss.
135
+ // Absence is not a miss, so it must not become 0.
136
+ export function readCacheUsage(usage) {
137
+ if (!usage || typeof usage !== "object") return null;
138
+ const out = {};
139
+ for (const key of CACHE_USAGE_FIELDS) {
140
+ const value = usage[key];
141
+ if (typeof value === "number" && Number.isFinite(value)) out[key] = value;
142
+ }
143
+ return Object.keys(out).length ? out : null;
144
+ }
145
+
146
+ export function applyCacheUsage(usage) {
147
+ const cache = readCacheUsage(usage);
148
+ if (!cache) return usage;
149
+ const cached =
150
+ typeof cache.cached_prompt_tokens === "number"
151
+ ? cache.cached_prompt_tokens
152
+ : cache.cache_read_input_tokens;
153
+ if (typeof cached !== "number") return usage;
154
+ const details =
155
+ usage.input_tokens_details && typeof usage.input_tokens_details === "object"
156
+ ? usage.input_tokens_details
157
+ : {};
158
+ if (typeof details.cached_tokens !== "number") {
159
+ details.cached_tokens = cached;
160
+ usage.input_tokens_details = details;
161
+ }
162
+ return usage;
163
+ }
package/src/proxy.mjs CHANGED
@@ -138,7 +138,7 @@ export async function openProxyStreamWithRetry(options, attempts = 2) {
138
138
  throw lastError;
139
139
  }
140
140
 
141
- export async function pipeProxySse(stream, output, map) {
141
+ export async function pipeProxySse(stream, output, map, usageBox) {
142
142
  const reader = stream.getReader();
143
143
  const decoder = new TextDecoder();
144
144
  let buffer = "";
@@ -166,5 +166,7 @@ export async function pipeProxySse(stream, output, map) {
166
166
  // unread holds the upstream socket open for the rest of the response.
167
167
  await reader.cancel(error).catch(() => {});
168
168
  throw error;
169
+ } finally {
170
+ if (usageBox) usageBox.cacheUsage = rewrite.cacheUsage ?? null;
169
171
  }
170
172
  }
package/src/tools.mjs CHANGED
@@ -5,6 +5,13 @@ import {
5
5
  saveGeneratedImage,
6
6
  } from "./imagegen.mjs";
7
7
  import { DEFAULT_GROK_MODEL, resolveGrokModel } from "./models.mjs";
8
+ import {
9
+ applyCacheUsage,
10
+ logicalFingerprint,
11
+ readCacheUsage,
12
+ stableJsonValue,
13
+ stableProxyName,
14
+ } from "./prefix.mjs";
8
15
 
9
16
  const OBJECT_SCHEMA = { type: "object", properties: {} };
10
17
 
@@ -227,7 +234,14 @@ function decodeCustomInput(argumentsValue) {
227
234
  }
228
235
 
229
236
  function addFunction(flattened, map, spec) {
230
- const proxyName = `codex_${flattened.length}_${sanitize(spec.name)}`;
237
+ // Index names change when Codex reorders tools, which rewrites every
238
+ // earlier function_call and breaks the prompt-cache prefix.
239
+ let proxyName = stableProxyName(spec, sanitize(spec.name));
240
+ if (map.has(proxyName)) {
241
+ let n = 2;
242
+ while (map.has(`${proxyName}_${n}`)) n += 1;
243
+ proxyName = `${proxyName}_${n}`;
244
+ }
231
245
  map.set(proxyName, {
232
246
  kind: spec.kind,
233
247
  namespace: spec.namespace,
@@ -239,10 +253,11 @@ function addFunction(flattened, map, spec) {
239
253
  description: spec.namespace
240
254
  ? `[${spec.namespace}] ${spec.description || spec.name}`
241
255
  : spec.description || spec.name,
242
- parameters:
256
+ parameters: stableJsonValue(
243
257
  spec.kind === "custom"
244
258
  ? structuredClone(CUSTOM_INPUT_SCHEMA)
245
259
  : usableParameters(spec.parameters),
260
+ ),
246
261
  });
247
262
  }
248
263
 
@@ -284,7 +299,7 @@ export function flattenCodexTools(tools = []) {
284
299
  }
285
300
  continue;
286
301
  }
287
- flattened.push(structuredClone(tool));
302
+ flattened.push(stableJsonValue(structuredClone(tool)));
288
303
  }
289
304
  return { tools: flattened, map };
290
305
  }
@@ -392,17 +407,36 @@ function toProxyInputNode(node, map, state) {
392
407
  return whitelistInputNode(next);
393
408
  }
394
409
 
395
- function toProxyInput(input, map) {
396
- const items = toProxyInputNode(input, map, { callIds: new Map() });
397
- if (!Array.isArray(items)) return items;
398
- return items.filter(
399
- (item) =>
400
- item &&
401
- typeof item === "object" &&
402
- (item.type == null || GROK_INPUT_ITEM_TYPES.has(item.type)),
410
+ function isForwardedItem(item) {
411
+ return (
412
+ item &&
413
+ item !== DROP &&
414
+ typeof item === "object" &&
415
+ !Array.isArray(item) &&
416
+ (item.type == null || GROK_INPUT_ITEM_TYPES.has(item.type))
403
417
  );
404
418
  }
405
419
 
420
+ function projectInput(input, map) {
421
+ if (!Array.isArray(input)) {
422
+ const items = toProxyInputNode(input, map, { callIds: new Map() });
423
+ return { items, pairs: [] };
424
+ }
425
+ const state = { callIds: new Map() };
426
+ const pairs = [];
427
+ for (const item of input) {
428
+ const fingerprint = logicalFingerprint(item);
429
+ const next = toProxyInputNode(item, map, state);
430
+ if (!isForwardedItem(next)) continue;
431
+ pairs.push({ fingerprint, item: next });
432
+ }
433
+ return { items: pairs.map((pair) => pair.item), pairs };
434
+ }
435
+
436
+ function toProxyInput(input, map) {
437
+ return projectInput(input, map).items;
438
+ }
439
+
406
440
  // Only the bridge knows a request is being served by the bridge. Codex does not
407
441
  // put the provider or model into the prompt, so without this line the model has
408
442
  // no way to answer "is Grok actually attached?" and either hedges or goes
@@ -432,9 +466,10 @@ export const IMAGE_GENERATION_PROVENANCE =
432
466
  export function toProxyRequest(body) {
433
467
  const { tools, map } = flattenCodexTools(body.tools ?? []);
434
468
  const effort = EFFORT[body.reasoning?.effort] ?? "high";
469
+ const projected = projectInput(body.input, map);
435
470
  const request = {
436
471
  model: resolveGrokModel(body.model),
437
- input: toProxyInput(body.input, map),
472
+ input: projected.items,
438
473
  tools,
439
474
  reasoning: { effort },
440
475
  stream: true,
@@ -463,7 +498,7 @@ export function toProxyRequest(body) {
463
498
  : provenance;
464
499
  if (typeof body.prompt_cache_key === "string" && body.prompt_cache_key)
465
500
  request.prompt_cache_key = body.prompt_cache_key;
466
- return { request, map };
501
+ return { request, map, projected: projected.pairs };
467
502
  }
468
503
 
469
504
  function rememberProxyItem(item, origin, state) {
@@ -508,8 +543,19 @@ function absorbGeneratedImage(item, state) {
508
543
  };
509
544
  }
510
545
 
546
+ const USAGE_EVENTS = new Set(["response.completed", "response.incomplete"]);
547
+
511
548
  function rewriteResponseEvent(value, map, state) {
512
549
  if (!value || typeof value !== "object") return value;
550
+ if (
551
+ USAGE_EVENTS.has(value.type) &&
552
+ value.response &&
553
+ typeof value.response.usage === "object"
554
+ ) {
555
+ applyCacheUsage(value.response.usage);
556
+ const cache = readCacheUsage(value.response.usage);
557
+ if (cache) state.cacheUsage = cache;
558
+ }
513
559
  if (
514
560
  typeof value.item === "object" &&
515
561
  value.item &&
@@ -569,6 +615,13 @@ export function createSseRewriter(map, options = {}) {
569
615
  callIds: new Map(),
570
616
  itemIds: new Map(),
571
617
  imageOptions: options.imageOptions,
618
+ cacheUsage: null,
572
619
  };
573
- return (block) => rewriteSseBlock(block, map, state);
620
+ const rewrite = (block) => rewriteSseBlock(block, map, state);
621
+ Object.defineProperty(rewrite, "cacheUsage", {
622
+ get() {
623
+ return state.cacheUsage;
624
+ },
625
+ });
626
+ return rewrite;
574
627
  }