@opengeni/codex 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/fetch.ts ADDED
@@ -0,0 +1,337 @@
1
+ // codexSubscriptionFetch — the transport installed on the OpenAI client for the
2
+ // "codex-subscription" provider. Mirrors the runtime's computerCallNormalizingFetch
3
+ // pattern: wraps a base fetch and returns a (input, init) => Promise<Response>.
4
+ //
5
+ // It reads the per-request Codex context from AsyncLocalStorage at CALL time, so a
6
+ // single process-cached client serves every workspace with the correct token. It:
7
+ // - rewrites /responses -> /codex/responses
8
+ // - injects the subscription auth headers (omits OpenAI-Beta on SSE; spec §1.2)
9
+ // - normalizes the request body (spec §0 verdict)
10
+ // - retries once on 401 after a forced token refresh (spec §1.9)
11
+ // Stream parsing is delegated to the SDK (SSE passthrough; spec §0(d)).
12
+
13
+ import { CODEX_ORIGINATOR } from "./constants";
14
+ import { normalizeCodexRequestBody } from "./normalize";
15
+ import { codexRequestStorage, type CodexTokenSnapshot, type CodexUsageHeaderSnapshot } from "./request-context";
16
+
17
+ export type FetchLike = (input: string | URL | Request, init?: RequestInit) => Promise<Response>;
18
+
19
+ /** Parse an integer header value; null when absent or not a finite integer. */
20
+ function parseIntHeader(value: string | null): number | null {
21
+ if (value === null) {
22
+ return null;
23
+ }
24
+ const n = Number.parseInt(value.trim(), 10);
25
+ return Number.isFinite(n) ? n : null;
26
+ }
27
+
28
+ /**
29
+ * Resolve a window reset instant from the response headers: prefer the absolute
30
+ * `*-reset-at` (epoch SECONDS → ms, mirroring codex-token-resolver's usage parse),
31
+ * else the relative `*-reset-after-seconds` from now, else now (a missing reset
32
+ * reads as "already cleared" — availableAt treats an elapsed reset as a bounded
33
+ * default cooldown, so the ranker never strands on it).
34
+ */
35
+ function resolveResetAt(headers: Headers, atKey: string, afterKey: string, nowMs: number): Date {
36
+ const at = parseIntHeader(headers.get(atKey));
37
+ if (at !== null) {
38
+ return new Date(at * 1000);
39
+ }
40
+ const after = parseIntHeader(headers.get(afterKey));
41
+ if (after !== null) {
42
+ return new Date(nowMs + after * 1000);
43
+ }
44
+ return new Date(nowMs);
45
+ }
46
+
47
+ /**
48
+ * Multi-account P4 (Part A): scrape the full usage snapshot the codex backend
49
+ * stamps on every `/codex/responses` response in `x-codex-primary-*` /
50
+ * `x-codex-secondary-*` headers (integer-identical to GET /wham/usage, for free).
51
+ *
52
+ * CRITICAL clobber-fix: return null unless BOTH windows expose a valid used-percent
53
+ * integer. recordCodexAccountUsage writes all five columns unconditionally, so a
54
+ * primary-only snapshot would null the weekly column. Both windows are always
55
+ * emitted together on `/codex/responses`; gating on both makes every write a full
56
+ * 5-column snapshot byte-identical to the poll path, and a malformed/absent header
57
+ * set simply no-ops (the /wham/usage poll fallback still covers it).
58
+ */
59
+ export function parseCodexUsageHeaders(headers: Headers): CodexUsageHeaderSnapshot | null {
60
+ const primaryUsedPercent = parseIntHeader(headers.get("x-codex-primary-used-percent"));
61
+ const secondaryUsedPercent = parseIntHeader(headers.get("x-codex-secondary-used-percent"));
62
+ if (primaryUsedPercent === null || secondaryUsedPercent === null) {
63
+ return null; // not a full both-windows snapshot — no-op (never a partial clobber)
64
+ }
65
+ const nowMs = Date.now();
66
+ return {
67
+ primaryUsedPercent,
68
+ primaryResetAt: resolveResetAt(headers, "x-codex-primary-reset-at", "x-codex-primary-reset-after-seconds", nowMs),
69
+ secondaryUsedPercent,
70
+ secondaryResetAt: resolveResetAt(headers, "x-codex-secondary-reset-at", "x-codex-secondary-reset-after-seconds", nowMs),
71
+ checkedAt: new Date(nowMs),
72
+ };
73
+ }
74
+
75
+ export function codexSubscriptionFetch(base: FetchLike = globalThis.fetch): FetchLike {
76
+ return async (input, init) => {
77
+ const ctx = codexRequestStorage.getStore();
78
+ if (!ctx) {
79
+ return base(input, init); // not a codex turn — passthrough, untouched
80
+ }
81
+
82
+ const rawUrl =
83
+ typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
84
+ // /responses -> /codex/responses, idempotent: the negative lookbehind skips
85
+ // URLs whose base already includes /codex (avoids /codex/codex/responses).
86
+ const rewritten = rawUrl.replace(/(?<!\/codex)\/responses(\b|$)/, "/codex/responses$1");
87
+
88
+ const attempt = async (auth: CodexTokenSnapshot): Promise<Response> => {
89
+ const headers = new Headers(init?.headers);
90
+ headers.set("Authorization", `Bearer ${auth.accessToken}`);
91
+ if (auth.chatgptAccountId) {
92
+ headers.set("ChatGPT-Account-ID", auth.chatgptAccountId);
93
+ }
94
+ headers.set("originator", CODEX_ORIGINATOR);
95
+ headers.set("User-Agent", `${CODEX_ORIGINATOR}/${ctx.clientVersion}`);
96
+ headers.set("version", ctx.clientVersion);
97
+ headers.set("accept", "text/event-stream");
98
+ headers.set("content-type", "application/json");
99
+ if (auth.isFedramp) {
100
+ headers.set("X-OpenAI-Fedramp", "true");
101
+ }
102
+ headers.delete("OpenAI-Beta"); // omit on SSE (spec §1.2); fallback: "responses=experimental" if backend 400s
103
+ headers.delete("x-api-key");
104
+
105
+ // The backend is streaming-only; force stream=true on the wire but remember
106
+ // the caller's intent so a non-streaming caller (e.g. the compaction
107
+ // summarizer) still gets a single JSON Response back.
108
+ let callerWantsStream = true;
109
+ const nextInit: RequestInit = { ...init, headers };
110
+ if (typeof init?.body === "string") {
111
+ try {
112
+ const parsed = JSON.parse(init.body) as Record<string, unknown>;
113
+ callerWantsStream = parsed.stream === true;
114
+ nextInit.body = JSON.stringify(normalizeCodexRequestBody(parsed, ctx.resolveModel));
115
+ } catch {
116
+ /* leave unparseable bodies untouched (already copied from init) */
117
+ }
118
+ }
119
+ if (process.env.CODEX_DEBUG) {
120
+ const keys = typeof nextInit.body === "string" ? Object.keys(JSON.parse(nextInit.body) as Record<string, unknown>) : [];
121
+ console.error(`[codex-debug] POST ${rewritten} stream=${callerWantsStream} bodyKeys=[${keys.join(",")}]`);
122
+ }
123
+ const res = await base(rewritten, nextInit);
124
+ // Multi-account P4 (Part A): scrape the usage headers ONCE, before the
125
+ // OK/!res.ok branch, so the same fire-and-forget read also covers the 429
126
+ // hard-cap path (an exhausted serving account stamps its own fresh
127
+ // used_percent with no extra fetch). Sync + non-throwing + never awaited;
128
+ // `if (usage)` makes an absent/malformed header set a safe no-op. We read
129
+ // res.headers only — the SSE body is never touched here.
130
+ const usage = parseCodexUsageHeaders(res.headers);
131
+ if (usage) {
132
+ ctx.onUsageHeaders?.(usage);
133
+ }
134
+ if (process.env.CODEX_DEBUG && !res.ok) {
135
+ console.error(`[codex-debug] <- ${res.status} ${await res.clone().text()}`);
136
+ }
137
+ // The codex backend leaves the terminal event's response.output empty and
138
+ // delivers the assistant items via output_item.done events instead. The
139
+ // @openai/agents parser (streaming AND non-streaming) reads response.output,
140
+ // so we must reconstruct it: collapse to one JSON Response for a non-streaming
141
+ // caller, or repair the live stream's terminal event for a streaming caller.
142
+ if (!res.ok) {
143
+ // Buffer the error body once and re-emit it as a concrete JSON Response.
144
+ // A streaming responses request whose error body is left as the raw
145
+ // (possibly SSE / already-streamed) Response makes the SDK throw
146
+ // "<status> status code (no body)" — the JSON error (type/message/
147
+ // resets_in_seconds) is lost, so a 429 usage cap surfaces as a generic,
148
+ // wrongly-retryable rate-limit. Re-emitting a clean application/json
149
+ // Response lets the SDK reconstruct error.error for EVERY codex error
150
+ // (401/400/5xx too). For a hard usage cap we also pin x-should-retry:false
151
+ // so the SDK does not burn its retry budget on a limit that won't lift.
152
+ return await bufferCodexErrorResponse(res);
153
+ }
154
+ return callerWantsStream ? repairCodexStream(res) : await sseToJsonResponse(res);
155
+ };
156
+
157
+ let res = await attempt(await ctx.getToken());
158
+ if (res.status === 401) {
159
+ res = await attempt(await ctx.refresh()); // single refresh-on-401 retry (spec §1.9)
160
+ }
161
+ return res;
162
+ };
163
+ }
164
+
165
+ /** The codex backend's hard-cap error type (ChatGPT/Codex usage limit reached). */
166
+ export const CODEX_USAGE_LIMIT_ERROR_TYPE = "usage_limit_reached";
167
+
168
+ export type CodexUsageLimitInfo = {
169
+ /** Seconds until the usage cap resets, when the backend reported it. */
170
+ resetsInSeconds: number | null;
171
+ };
172
+
173
+ /**
174
+ * Classify a thrown error as a ChatGPT/Codex usage-cap (429 usage_limit_reached)
175
+ * and extract the reset window. The SDK surfaces the codex backend's 429 as an
176
+ * OpenAI APIError whose `.type` (and `.error.type`) is `usage_limit_reached` and
177
+ * whose `.error.resets_in_seconds` carries the cap reset. Walks the cause chain
178
+ * and tolerates the message-only shape so it survives any SDK re-wrapping.
179
+ * Returns null for anything that is not a usage cap.
180
+ */
181
+ export function classifyCodexUsageLimitError(error: unknown): CodexUsageLimitInfo | null {
182
+ let cur: unknown = error;
183
+ for (let depth = 0; depth < 6 && cur && typeof cur === "object"; depth++) {
184
+ const e = cur as Record<string, unknown>;
185
+ const body = (e.error && typeof e.error === "object" ? e.error : undefined) as Record<string, unknown> | undefined;
186
+ const type = (typeof e.type === "string" ? e.type : undefined) ?? (typeof body?.type === "string" ? body.type : undefined);
187
+ const message = typeof e.message === "string" ? e.message : "";
188
+ const status = Number(e.status);
189
+ if (
190
+ type === CODEX_USAGE_LIMIT_ERROR_TYPE ||
191
+ message.includes(CODEX_USAGE_LIMIT_ERROR_TYPE) ||
192
+ (status === 429 && /usage limit/i.test(message))
193
+ ) {
194
+ const resets =
195
+ (typeof body?.resets_in_seconds === "number" ? body.resets_in_seconds : undefined) ??
196
+ (typeof e.resets_in_seconds === "number" ? (e.resets_in_seconds as number) : undefined) ??
197
+ null;
198
+ return { resetsInSeconds: resets };
199
+ }
200
+ cur = e.cause;
201
+ }
202
+ return null;
203
+ }
204
+
205
+ /**
206
+ * Buffer a non-OK codex Response and re-emit it as a clean `application/json`
207
+ * Response so the SDK can reconstruct `error.error` from the body. A 429 usage
208
+ * cap (`error.type === "usage_limit_reached"`) is a HARD limit, not transient
209
+ * backpressure, so we pin `x-should-retry: false` to stop the SDK retrying it.
210
+ * Reading the body here also drains the socket of a discarded 401 (no leak).
211
+ */
212
+ async function bufferCodexErrorResponse(res: Response): Promise<Response> {
213
+ const bodyText = await res.text().catch(() => "");
214
+ const headers = new Headers(res.headers);
215
+ headers.set("content-type", "application/json");
216
+ headers.delete("content-length"); // body re-serialized
217
+ headers.delete("content-encoding"); // text() already decoded any gzip
218
+ let errorType: string | undefined;
219
+ try {
220
+ const parsed = JSON.parse(bodyText) as { error?: { type?: unknown } };
221
+ errorType = typeof parsed.error?.type === "string" ? parsed.error.type : undefined;
222
+ } catch {
223
+ /* non-JSON error body — leave as-is, no retry-header override */
224
+ }
225
+ if (errorType === CODEX_USAGE_LIMIT_ERROR_TYPE) {
226
+ headers.set("x-should-retry", "false");
227
+ }
228
+ return new Response(bodyText, { status: res.status, statusText: res.statusText, headers });
229
+ }
230
+
231
+ /**
232
+ * Collapse a Responses SSE stream into the single JSON Response object a
233
+ * non-streaming `responses.create` caller expects: the terminal response.*
234
+ * event carries the full `response` payload.
235
+ */
236
+ async function sseToJsonResponse(res: Response): Promise<Response> {
237
+ const text = await res.text();
238
+ let final: Record<string, unknown> | null = null;
239
+ const items: unknown[] = []; // assembled from output_item.done (the codex backend
240
+ // leaves response.completed.response.output empty and emits the items separately).
241
+ for (const block of text.split("\n\n")) {
242
+ const data = block
243
+ .split("\n")
244
+ .filter((l) => l.startsWith("data:"))
245
+ .map((l) => l.slice(5).trim())
246
+ .join("\n");
247
+ if (!data || data === "[DONE]") {
248
+ continue;
249
+ }
250
+ try {
251
+ const ev = JSON.parse(data) as { type?: string; response?: Record<string, unknown>; item?: unknown };
252
+ if (ev.type === "response.output_item.done" && ev.item !== undefined) {
253
+ items.push(ev.item);
254
+ } else if (ev.type === "response.completed" || ev.type === "response.done" || ev.type === "response.incomplete") {
255
+ final = ev.response ?? null;
256
+ }
257
+ } catch {
258
+ /* ignore non-JSON keepalive lines */
259
+ }
260
+ }
261
+ if (final && items.length > 0) {
262
+ final = { ...final, output: items }; // prefer the assembled items over an empty output array
263
+ }
264
+ if (process.env.CODEX_DEBUG) {
265
+ console.error(`[codex-debug] sse->json items=${items.length} outputLen=${Array.isArray(final?.output) ? (final.output as unknown[]).length : "?"}`);
266
+ }
267
+ const headers = new Headers(res.headers);
268
+ headers.set("content-type", "application/json");
269
+ headers.delete("content-length");
270
+ return new Response(JSON.stringify(final ?? {}), { status: 200, headers });
271
+ }
272
+
273
+ /**
274
+ * Repair a live Responses SSE stream for the @openai/agents streaming parser: pass
275
+ * every event through unchanged, collect the output_item.done items, and inject
276
+ * them into the terminal event's empty `output` so the parser sees the message.
277
+ */
278
+ function repairCodexStream(res: Response): Response {
279
+ if (!res.body) {
280
+ return res;
281
+ }
282
+ const items: unknown[] = [];
283
+ const decoder = new TextDecoder();
284
+ const encoder = new TextEncoder();
285
+ let buffer = "";
286
+ const transform = new TransformStream<Uint8Array, Uint8Array>({
287
+ transform(chunk, controller) {
288
+ buffer += decoder.decode(chunk, { stream: true });
289
+ let idx = buffer.indexOf("\n\n");
290
+ while (idx !== -1) {
291
+ const block = buffer.slice(0, idx);
292
+ buffer = buffer.slice(idx + 2);
293
+ controller.enqueue(encoder.encode(`${patchSseBlock(block, items)}\n\n`));
294
+ idx = buffer.indexOf("\n\n");
295
+ }
296
+ },
297
+ flush(controller) {
298
+ if (buffer.length > 0) {
299
+ controller.enqueue(encoder.encode(patchSseBlock(buffer, items)));
300
+ }
301
+ },
302
+ });
303
+ const headers = new Headers(res.headers);
304
+ headers.delete("content-length");
305
+ return new Response(res.body.pipeThrough(transform), { status: res.status, headers });
306
+ }
307
+
308
+ /** Collect output_item.done items (mutating `items`); rewrite the terminal event's empty output. */
309
+ function patchSseBlock(block: string, items: unknown[]): string {
310
+ const lines = block.split("\n");
311
+ const dataStr = lines.filter((l) => l.startsWith("data:")).map((l) => l.slice(5).trim()).join("\n");
312
+ if (!dataStr || dataStr === "[DONE]") {
313
+ return block;
314
+ }
315
+ let ev: { type?: string; item?: unknown; response?: Record<string, unknown> };
316
+ try {
317
+ ev = JSON.parse(dataStr);
318
+ } catch {
319
+ return block;
320
+ }
321
+ if (ev.type === "response.output_item.done" && ev.item !== undefined) {
322
+ items.push(ev.item);
323
+ return block;
324
+ }
325
+ if (
326
+ (ev.type === "response.completed" || ev.type === "response.done" || ev.type === "response.incomplete") &&
327
+ ev.response
328
+ ) {
329
+ const out = ev.response.output;
330
+ if ((!Array.isArray(out) || out.length === 0) && items.length > 0) {
331
+ ev.response = { ...ev.response, output: items };
332
+ const nonData = lines.filter((l) => !l.startsWith("data:"));
333
+ return [...nonData, `data: ${JSON.stringify(ev)}`].join("\n");
334
+ }
335
+ }
336
+ return block;
337
+ }
package/src/index.ts ADDED
@@ -0,0 +1,10 @@
1
+ export * from "./constants";
2
+ export * from "./billing";
3
+ export * from "./device-code";
4
+ export * from "./refresh";
5
+ export * from "./normalize";
6
+ export * from "./usage-normalize";
7
+ export * from "./api-client";
8
+ export * from "./request-context";
9
+ export * from "./fetch";
10
+ export * from "./mcp-sanitize";
@@ -0,0 +1,290 @@
1
+ // The codex_apps connector MCP is incompatible with the Responses API tool
2
+ // contract in two ways that each fail the whole turn:
3
+ //
4
+ // 1. NAMES. Connector tools are named like "vercel.deploy_to_vercel" (dots).
5
+ // The Responses API requires every function-tool name to match
6
+ // ^[A-Za-z0-9_-]+$, so the request 400s ("Invalid 'tools[0].name': string
7
+ // does not match pattern"). We cannot just rename them in tools/list — the
8
+ // model would then call a name the MCP server does not know. So we remap
9
+ // BIDIRECTIONALLY at the transport: sanitize the name (and remember the
10
+ // mapping) on the tools/list RESPONSE, and reverse it back to the original on
11
+ // the tools/call REQUEST.
12
+ //
13
+ // 2. OUTPUT SCHEMAS. 122 of 217 tools return an empty `outputSchema: {}` (no
14
+ // `type`). @modelcontextprotocol/sdk validates every tool's outputSchema as a
15
+ // strict `{ type: "object", ... }` and ZodErrors the WHOLE tools/list. Since
16
+ // codex_apps runs with cacheToolsList:false it re-lists per turn, so that
17
+ // error (thrown during tool enumeration, outside the best-effort connect
18
+ // wrapper) fails the turn. We drop any non-object outputSchema before the
19
+ // validator sees it — safe, as outputSchema is an advisory hint only.
20
+
21
+ import { CODEX_APPS_MCP_SERVER_ID } from "./constants";
22
+ import type { FetchLike } from "./fetch";
23
+
24
+ const VALID_TOOL_NAME = /^[a-zA-Z0-9_-]+$/;
25
+
26
+ // The Responses API rejects a function-tool name longer than 64 chars (it 400s
27
+ // the WHOLE turn). Some namespaced connector tool names exceed this, and the
28
+ // collision-disambiguation suffix only lengthens names, so the mapper must cap
29
+ // length too — not just charset.
30
+ const MAX_TOOL_NAME_LEN = 64;
31
+
32
+ // CRITICAL: this sanitizer runs on the codex_apps tools/list wire BEFORE OpenGeni's
33
+ // PrefixedMcpServer (packages/runtime) prepends `<serverId>__` to every tool name
34
+ // (prefixedMcpToolName). The 64-char limit applies to that FINAL prefixed name the
35
+ // model sees, so a name we cap at 64 here becomes 64 + 12 = 76 after prefixing and
36
+ // 400s the whole turn. Reserve the runtime prefix so `codex_apps__<sanitized>` is
37
+ // always <= 64. The sanitizer owns the server id, so the reservation is exact and
38
+ // stays self-contained (no runtime import). The reverse mapping is unaffected: the
39
+ // mapper is keyed on the pre-prefix sanitized name, which is what tools/call carries
40
+ // back after PrefixedMcpServer strips its prefix.
41
+ const RUNTIME_TOOL_NAME_PREFIX_LEN = CODEX_APPS_MCP_SERVER_ID.length + "__".length; // `codex_apps__` = 12
42
+ const EFFECTIVE_MAX_TOOL_NAME_LEN = MAX_TOOL_NAME_LEN - RUNTIME_TOOL_NAME_PREFIX_LEN; // 52
43
+
44
+ /** Short, stable, charset-legal hash of a string (djb2 → base36). Deterministic. */
45
+ function shortHash(input: string): string {
46
+ let h = 5381;
47
+ for (let i = 0; i < input.length; i++) {
48
+ h = ((h << 5) + h + input.charCodeAt(i)) >>> 0; // h * 33 + c, kept unsigned
49
+ }
50
+ return h.toString(36);
51
+ }
52
+
53
+ /** Truncate to <= EFFECTIVE_MAX_TOOL_NAME_LEN (reserving the runtime prefix), appending `_<hash(original)>` so the result stays unique + deterministic. */
54
+ function capLength(candidate: string, original: string): string {
55
+ if (candidate.length <= EFFECTIVE_MAX_TOOL_NAME_LEN) {
56
+ return candidate;
57
+ }
58
+ const suffix = `_${shortHash(original)}`;
59
+ return candidate.slice(0, Math.max(0, EFFECTIVE_MAX_TOOL_NAME_LEN - suffix.length)) + suffix;
60
+ }
61
+
62
+ /**
63
+ * Maps connector tool names to a Responses-API-legal charset and back. One
64
+ * instance per codex_apps transport (i.e. per turn): tools/list populates it,
65
+ * tools/call reads it. Idempotent across repeat listings.
66
+ */
67
+ export class ToolNameMapper {
68
+ private readonly sanitizedToOriginal = new Map<string, string>();
69
+ private readonly used = new Set<string>();
70
+
71
+ /** Return a legal, unique name (<= EFFECTIVE_MAX_TOOL_NAME_LEN, so `<prefix>__name` <= 64) for `original`, recording the reverse mapping. */
72
+ sanitize(original: string): string {
73
+ let candidate = VALID_TOOL_NAME.test(original)
74
+ ? original
75
+ : original.replace(/[^a-zA-Z0-9_-]/g, "_") || "tool";
76
+ // Enforce the Responses-API 64-char cap (stable hash suffix keyed on the
77
+ // ORIGINAL → deterministic across repeat listings, distinct originals don't
78
+ // collide after truncation).
79
+ candidate = capLength(candidate, original);
80
+ // Disambiguate a genuine collision with a DIFFERENT original (never with
81
+ // the same original — that keeps repeat listings stable/idempotent). Re-cap
82
+ // after each suffix so disambiguation never re-breaches the effective limit.
83
+ if (this.used.has(candidate) && this.sanitizedToOriginal.get(candidate) !== original) {
84
+ const base = candidate;
85
+ let n = 2;
86
+ do {
87
+ const suffix = `_${n++}`;
88
+ candidate = (base.length + suffix.length > EFFECTIVE_MAX_TOOL_NAME_LEN
89
+ ? base.slice(0, EFFECTIVE_MAX_TOOL_NAME_LEN - suffix.length)
90
+ : base) + suffix;
91
+ } while (this.used.has(candidate));
92
+ }
93
+ this.used.add(candidate);
94
+ this.sanitizedToOriginal.set(candidate, original);
95
+ return candidate;
96
+ }
97
+
98
+ /** Reverse a sanitized name back to the MCP server's original, if known. */
99
+ toOriginal(sanitized: string): string | undefined {
100
+ return this.sanitizedToOriginal.get(sanitized);
101
+ }
102
+ }
103
+
104
+ /**
105
+ * Drop bad outputSchemas + sanitize tool names on a JSON-RPC tools/list result, in place.
106
+ *
107
+ * P4 (Part B.1): when `namespaceSink` is provided, accumulate each tool's ORIGINAL
108
+ * connector namespace (the segment BEFORE the first dot, e.g. `github` from
109
+ * `github.create_issue`) into it — captured HERE because this pass sees the original
110
+ * dotted name BEFORE mapper.sanitize rewrites the dot away. Only dotted names carry a
111
+ * connector namespace; un-dotted (already-legal) names are not connectors and are skipped.
112
+ */
113
+ function sanitizeToolsInRpcMessage(message: unknown, mapper: ToolNameMapper, namespaceSink?: Set<string>): void {
114
+ if (!message || typeof message !== "object") {
115
+ return;
116
+ }
117
+ const tools = (message as { result?: { tools?: unknown } }).result?.tools;
118
+ if (!Array.isArray(tools)) {
119
+ return;
120
+ }
121
+ for (const tool of tools) {
122
+ if (!tool || typeof tool !== "object") {
123
+ continue;
124
+ }
125
+ const record = tool as Record<string, unknown>;
126
+ if ("outputSchema" in record) {
127
+ // Drop EVERY outputSchema, not just malformed/empty ones. The MCP SDK client
128
+ // caches a validator for any tool that declares an outputSchema and validates
129
+ // each tool CALL's `structuredContent` against it — and the codex_apps
130
+ // connectors return results that do NOT match their own declared schemas
131
+ // (e.g. the schema requires a `result` property the response omits), so the
132
+ // SDK throws `McpError -32602: Structured content does not match the tool's
133
+ // output schema` and EVERY such connector tool call fails (observed live:
134
+ // gmail_search_emails / gmail_get_profile / gmail_list_labels all -32602ed).
135
+ // outputSchema is advisory — the agent reads the text `content` regardless —
136
+ // so dropping it makes the connector tools usable. This also subsumes the
137
+ // empty-`{}` case the strict Tool schema rejected at tools/list time.
138
+ delete record.outputSchema;
139
+ }
140
+ if (typeof record.name === "string") {
141
+ if (namespaceSink && record.name.includes(".")) {
142
+ const namespace = record.name.slice(0, record.name.indexOf("."));
143
+ if (namespace) {
144
+ namespaceSink.add(namespace);
145
+ }
146
+ }
147
+ record.name = mapper.sanitize(record.name);
148
+ }
149
+ }
150
+ }
151
+
152
+ /**
153
+ * Surface a tool CALL's `structuredContent` to the model by inlining it as a text
154
+ * `content` block, in place.
155
+ *
156
+ * WHY. The @openai/agents MCP bridge forwards ONLY `result.content` to the model
157
+ * (agents-core shims/mcp-server: `const result = parsed.content`) and DISCARDS
158
+ * `result.structuredContent`. The codex_apps connectors return the real payload in
159
+ * `structuredContent` and a bare `"Action completed."` placeholder in `content`
160
+ * (verified live: `gmail.get_profile` → content=[{text:"Action completed."}],
161
+ * structuredContent={id,name,email,…}). Without this the agent's tool call
162
+ * "succeeds" but carries NO data — the model sees only the placeholder. Appending
163
+ * the structured payload as a text block makes the data reach the model while
164
+ * leaving the original content untouched.
165
+ *
166
+ * No-op when there is no `structuredContent` — so a tools/list response (or any
167
+ * result without it) passes through unchanged. Runs alongside (after) the empty
168
+ * outputSchema drop, which is what stops the MCP SDK -32602ing this same result.
169
+ */
170
+ function inlineStructuredContentInRpcMessage(message: unknown): void {
171
+ if (!message || typeof message !== "object") {
172
+ return;
173
+ }
174
+ const result = (message as { result?: unknown }).result;
175
+ if (!result || typeof result !== "object") {
176
+ return;
177
+ }
178
+ const record = result as Record<string, unknown>;
179
+ if (!("structuredContent" in record)) {
180
+ return;
181
+ }
182
+ const structured = record.structuredContent;
183
+ if (structured === undefined || structured === null) {
184
+ return;
185
+ }
186
+ const text = typeof structured === "string" ? structured : JSON.stringify(structured);
187
+ const content = Array.isArray(record.content) ? [...record.content] : [];
188
+ content.push({ type: "text", text });
189
+ record.content = content;
190
+ }
191
+
192
+ /** Sanitize a single JSON body (application/json MCP response). */
193
+ export function sanitizeMcpJsonBody(text: string, mapper: ToolNameMapper = new ToolNameMapper(), namespaceSink?: Set<string>): string {
194
+ try {
195
+ const parsed = JSON.parse(text);
196
+ sanitizeToolsInRpcMessage(parsed, mapper, namespaceSink);
197
+ inlineStructuredContentInRpcMessage(parsed);
198
+ return JSON.stringify(parsed);
199
+ } catch {
200
+ return text; // not JSON we understand — leave untouched
201
+ }
202
+ }
203
+
204
+ /** Sanitize an SSE body: each JSON-RPC message rides on a `data:` line. */
205
+ export function sanitizeMcpSseBody(text: string, mapper: ToolNameMapper = new ToolNameMapper(), namespaceSink?: Set<string>): string {
206
+ return text
207
+ .split("\n")
208
+ .map((line) => {
209
+ if (!line.startsWith("data:")) {
210
+ return line;
211
+ }
212
+ const payload = line.slice("data:".length).trimStart();
213
+ try {
214
+ const parsed = JSON.parse(payload);
215
+ sanitizeToolsInRpcMessage(parsed, mapper, namespaceSink);
216
+ inlineStructuredContentInRpcMessage(parsed);
217
+ return `data: ${JSON.stringify(parsed)}`;
218
+ } catch {
219
+ return line;
220
+ }
221
+ })
222
+ .join("\n");
223
+ }
224
+
225
+ /** Reverse a sanitized tools/call name back to the original; returns null if no rewrite is needed. */
226
+ export function remapToolCallRequestBody(body: string, mapper: ToolNameMapper): string | null {
227
+ try {
228
+ const message = JSON.parse(body) as { method?: unknown; params?: { name?: unknown } };
229
+ if (message.method !== "tools/call") {
230
+ return null;
231
+ }
232
+ const name = message.params?.name;
233
+ if (typeof name !== "string") {
234
+ return null;
235
+ }
236
+ const original = mapper.toOriginal(name);
237
+ if (original === undefined || original === name) {
238
+ return null;
239
+ }
240
+ message.params!.name = original;
241
+ return JSON.stringify(message);
242
+ } catch {
243
+ return null;
244
+ }
245
+ }
246
+
247
+ /**
248
+ * Wrap a base fetch so the codex_apps MCP transport is Responses-API-compatible:
249
+ * tools/list responses get their names sanitized + bad outputSchemas dropped (and
250
+ * the name mapping recorded), and tools/call requests get their name reversed back
251
+ * to the MCP server's original. Only the POST request/response is buffered; the
252
+ * long-lived GET notification SSE stream is passed through untouched.
253
+ *
254
+ * P4 (Part B.1): an optional `namespaceSink` Set accumulates the ORIGINAL-dotted
255
+ * connector namespaces seen across every tools/list this turn (captured before the
256
+ * dot is sanitized away). The worker reads the (live, by-reference) Set after the
257
+ * turn to cache the serving account's connector set — packages/codex stays db-free.
258
+ */
259
+ export function codexAppsSanitizingFetch(base: FetchLike = globalThis.fetch, namespaceSink?: Set<string>): FetchLike {
260
+ const mapper = new ToolNameMapper();
261
+ return async (input, init) => {
262
+ // Outgoing: reverse a sanitized tools/call name to the server's original.
263
+ let nextInit = init;
264
+ if (init && typeof init.body === "string" && (init.method ?? "GET").toUpperCase() === "POST") {
265
+ const remapped = remapToolCallRequestBody(init.body, mapper);
266
+ if (remapped !== null) {
267
+ nextInit = { ...init, body: remapped };
268
+ }
269
+ }
270
+ const res = await base(input, nextInit);
271
+ const method = (init?.method ?? (input instanceof Request ? input.method : "GET")).toUpperCase();
272
+ if (method !== "POST" || !res.ok || !res.body) {
273
+ return res;
274
+ }
275
+ const contentType = res.headers.get("content-type") ?? "";
276
+ const isJson = contentType.includes("application/json");
277
+ const isSse = contentType.includes("text/event-stream");
278
+ if (!isJson && !isSse) {
279
+ return res;
280
+ }
281
+ const originalBody = await res.text();
282
+ const sanitized = isJson
283
+ ? sanitizeMcpJsonBody(originalBody, mapper, namespaceSink)
284
+ : sanitizeMcpSseBody(originalBody, mapper, namespaceSink);
285
+ const headers = new Headers(res.headers);
286
+ headers.delete("content-length"); // body length changed
287
+ headers.delete("content-encoding");
288
+ return new Response(sanitized, { status: res.status, statusText: res.statusText, headers });
289
+ };
290
+ }