@moxxy/plugin-provider-openai-codex 0.41.2 → 0.42.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,169 @@
1
+ import { once } from 'node:events';
2
+ import { createServer, type IncomingHttpHeaders, type ServerResponse } from 'node:http';
3
+ import type { AddressInfo } from 'node:net';
4
+ import { afterEach, describe, expect, it } from 'vitest';
5
+ import {
6
+ GPT_LIVE_MODEL,
7
+ GptLiveCallClient,
8
+ buildGptLiveCallsUrl,
9
+ } from './gpt-live-call.js';
10
+
11
+ // The ChatGPT realtime backend is an external network API, so these tests stand
12
+ // up a real loopback HTTP server in its place (the same seam the Codex
13
+ // transcriber uses). Everything on the moxxy side — URL, headers, body, SDP and
14
+ // Location parsing — runs for real.
15
+ interface Captured {
16
+ readonly method: string;
17
+ readonly url: string;
18
+ readonly headers: IncomingHttpHeaders;
19
+ readonly body: unknown;
20
+ }
21
+
22
+ const OFFER = 'v=0\r\no=- 1 2 IN IP4 127.0.0.1\r\ns=-\r\nt=0 0\r\n';
23
+ const ANSWER = 'v=0\r\no=- 3 4 IN IP4 0.0.0.0\r\ns=-\r\nt=0 0\r\na=ice-lite\r\n';
24
+ const closers: Array<() => Promise<void>> = [];
25
+
26
+ afterEach(async () => {
27
+ await Promise.all(closers.splice(0).map((close) => close()));
28
+ });
29
+
30
+ async function startBackend(
31
+ respond: (res: ServerResponse) => void,
32
+ ): Promise<{ readonly baseUrl: string; readonly requests: Captured[] }> {
33
+ const requests: Captured[] = [];
34
+ const server = createServer(async (req, res) => {
35
+ const chunks: Buffer[] = [];
36
+ for await (const chunk of req) chunks.push(Buffer.from(chunk as Buffer));
37
+ const raw = Buffer.concat(chunks).toString('utf8');
38
+ requests.push({
39
+ method: req.method ?? '',
40
+ url: req.url ?? '',
41
+ headers: req.headers,
42
+ body: raw ? JSON.parse(raw) : null,
43
+ });
44
+ respond(res);
45
+ });
46
+ server.listen(0, '127.0.0.1');
47
+ await once(server, 'listening');
48
+ closers.push(
49
+ () =>
50
+ new Promise((resolve) => {
51
+ server.closeAllConnections();
52
+ server.close(() => resolve());
53
+ }),
54
+ );
55
+ const { port } = server.address() as AddressInfo;
56
+ return { baseUrl: `http://127.0.0.1:${port}/backend-api`, requests };
57
+ }
58
+
59
+ function answerWith(res: ServerResponse): void {
60
+ res.writeHead(201, {
61
+ 'content-type': 'text/plain; charset=utf-8',
62
+ location: '/v1/realtime/calls/rtc_u32_ETCRZp3oN5zeME7AqoW9SLcJTYbQ5Vvy',
63
+ });
64
+ res.end(ANSWER);
65
+ }
66
+
67
+ function client(baseUrl: string): GptLiveCallClient {
68
+ return new GptLiveCallClient({
69
+ baseUrl,
70
+ sessionIdProvider: () => 'voice-session-1',
71
+ resolveCredentials: async () => ({ accessToken: 'access-token-123', accountId: 'acct-9' }),
72
+ });
73
+ }
74
+
75
+ describe('GptLiveCallClient', () => {
76
+ it('negotiates a GPT-Live WebRTC call with the existing ChatGPT OAuth credentials', async () => {
77
+ const backend = await startBackend(answerWith);
78
+
79
+ const answer = await client(backend.baseUrl).start({
80
+ sdp: OFFER,
81
+ instructions: 'Answer the user yourself.',
82
+ history: [
83
+ { role: 'user', text: 'Sekretne słowo to pomarańcza.' },
84
+ { role: 'assistant', text: 'Zapamiętałem.' },
85
+ ],
86
+ });
87
+
88
+ expect(answer).toEqual({ sdp: ANSWER, callId: 'rtc_u32_ETCRZp3oN5zeME7AqoW9SLcJTYbQ5Vvy' });
89
+ const [request] = backend.requests;
90
+ expect(request?.method).toBe('POST');
91
+ expect(request?.url).toBe('/backend-api/codex/realtime/calls?intent=quicksilver&architecture=avas');
92
+ expect(request?.headers).toMatchObject({
93
+ authorization: 'Bearer access-token-123',
94
+ 'chatgpt-account-id': 'acct-9',
95
+ 'openai-alpha': 'quicksilver=v2',
96
+ originator: 'codex_cli_rs',
97
+ 'x-session-id': 'voice-session-1',
98
+ accept: 'application/sdp',
99
+ 'content-type': 'application/json',
100
+ });
101
+ expect(request?.body).toEqual({
102
+ sdp: OFFER,
103
+ session: {
104
+ instructions: 'Answer the user yourself.',
105
+ model: GPT_LIVE_MODEL,
106
+ audio: { output: { voice: 'maple' } },
107
+ delegation: { type: 'client' },
108
+ initial_items: [
109
+ { type: 'message', role: 'user', content: [{ type: 'input_text', text: 'Sekretne słowo to pomarańcza.' }] },
110
+ { type: 'message', role: 'assistant', content: [{ type: 'output_text', text: 'Zapamiętałem.' }] },
111
+ ],
112
+ },
113
+ });
114
+ });
115
+
116
+ it('omits initial items for an empty conversation', async () => {
117
+ const backend = await startBackend(answerWith);
118
+
119
+ await client(backend.baseUrl).start({ sdp: OFFER, instructions: 'Hi.', history: [] });
120
+
121
+ const body = backend.requests[0]?.body as { session: Record<string, unknown> };
122
+ expect(body.session).not.toHaveProperty('initial_items');
123
+ });
124
+
125
+ it('surfaces the backend rejection without leaking the bearer token', async () => {
126
+ const backend = await startBackend((res) => {
127
+ res.writeHead(403, { 'content-type': 'application/json' });
128
+ res.end('{"detail":"Voice session access denied"}');
129
+ });
130
+
131
+ const failure = client(backend.baseUrl).start({ sdp: OFFER, instructions: 'x', history: [] });
132
+
133
+ await expect(failure).rejects.toThrow(/403.*Voice session access denied/);
134
+ await expect(failure).rejects.not.toThrow(/access-token-123/);
135
+ });
136
+
137
+ it('rejects an answer without a call location', async () => {
138
+ const backend = await startBackend((res) => {
139
+ res.writeHead(201, { 'content-type': 'text/plain' });
140
+ res.end(ANSWER);
141
+ });
142
+
143
+ await expect(
144
+ client(backend.baseUrl).start({ sdp: OFFER, instructions: 'x', history: [] }),
145
+ ).rejects.toThrow(/Location/);
146
+ });
147
+
148
+ it('rejects an invalid offer before any credential leaves the process', async () => {
149
+ const backend = await startBackend(answerWith);
150
+
151
+ await expect(
152
+ client(backend.baseUrl).start({ sdp: 'not-an-sdp', instructions: 'x', history: [] }),
153
+ ).rejects.toThrow(/offer/);
154
+ expect(backend.requests).toHaveLength(0);
155
+ });
156
+ });
157
+
158
+ describe('buildGptLiveCallsUrl', () => {
159
+ it('targets the ChatGPT backend by default', () => {
160
+ expect(buildGptLiveCallsUrl()).toBe(
161
+ 'https://chatgpt.com/backend-api/codex/realtime/calls?intent=quicksilver&architecture=avas',
162
+ );
163
+ });
164
+
165
+ it('refuses to send OAuth credentials to any other origin', () => {
166
+ expect(() => buildGptLiveCallsUrl('https://evil.example/backend-api')).toThrow(/Refusing/);
167
+ expect(() => buildGptLiveCallsUrl('http://chatgpt.com/backend-api')).toThrow(/Refusing/);
168
+ });
169
+ });
@@ -0,0 +1,148 @@
1
+ import { randomUUID } from 'node:crypto';
2
+ import { ORIGINATOR } from '../oauth.js';
3
+ import type { GptLiveHistoryItem } from './gpt-live-history.js';
4
+
5
+ /**
6
+ * GPT-Live over the user's existing ChatGPT OAuth login.
7
+ *
8
+ * The public `/v1/live/sessions` API requires a platform API key; a ChatGPT
9
+ * subscription reaches GPT-Live only through the Codex realtime route below —
10
+ * the same one the open-source Codex app-server uses. It is an undocumented
11
+ * preview contract, so every wire detail lives in this one module.
12
+ */
13
+ export const DEFAULT_GPT_LIVE_BASE_URL = 'https://chatgpt.com/backend-api';
14
+ export const GPT_LIVE_MODEL = 'gpt-live-1-codex';
15
+ export const GPT_LIVE_DEFAULT_VOICE = 'maple';
16
+ const GPT_LIVE_PROTOCOL = 'quicksilver=v2';
17
+ const CALLS_PATH = '/codex/realtime/calls';
18
+ const CALLS_QUERY = '?intent=quicksilver&architecture=avas';
19
+ const DEFAULT_TIMEOUT_MS = 30_000;
20
+ const MAX_SDP_BYTES = 1_000_000;
21
+ const MAX_ERROR_CHARS = 2_048;
22
+ const CALL_ID = /^(rtc_[A-Za-z0-9_-]+|[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})$/i;
23
+
24
+ export interface GptLiveCredentials {
25
+ readonly accessToken: string;
26
+ readonly accountId?: string;
27
+ }
28
+
29
+ export interface GptLiveCallStart {
30
+ /** The browser-created WebRTC offer. */
31
+ readonly sdp: string;
32
+ readonly instructions: string;
33
+ /** Conversation the model should already know when the call opens. */
34
+ readonly history: ReadonlyArray<GptLiveHistoryItem>;
35
+ readonly voice?: string;
36
+ }
37
+
38
+ export interface GptLiveCallAnswer {
39
+ readonly sdp: string;
40
+ readonly callId: string;
41
+ }
42
+
43
+ export interface GptLiveCallClientOptions {
44
+ readonly resolveCredentials: () => Promise<GptLiveCredentials>;
45
+ /** `https://chatgpt.com/backend-api`, or a loopback host for tests. */
46
+ readonly baseUrl?: string;
47
+ readonly sessionIdProvider?: () => string;
48
+ readonly timeoutMs?: number;
49
+ }
50
+
51
+ /** Negotiates one GPT-Live WebRTC call; audio and events then flow peer-to-peer. */
52
+ export class GptLiveCallClient {
53
+ private readonly resolveCredentials: () => Promise<GptLiveCredentials>;
54
+ private readonly callsUrl: string;
55
+ private readonly sessionIdProvider: () => string;
56
+ private readonly timeoutMs: number;
57
+
58
+ constructor(options: GptLiveCallClientOptions) {
59
+ this.resolveCredentials = options.resolveCredentials;
60
+ this.callsUrl = buildGptLiveCallsUrl(options.baseUrl);
61
+ this.sessionIdProvider = options.sessionIdProvider ?? randomUUID;
62
+ this.timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS;
63
+ }
64
+
65
+ async start(input: GptLiveCallStart): Promise<GptLiveCallAnswer> {
66
+ assertSdp(input.sdp, 'offer');
67
+ const credentials = await this.resolveCredentials();
68
+ const headers: Record<string, string> = {
69
+ 'Content-Type': 'application/json',
70
+ Accept: 'application/sdp',
71
+ Authorization: `Bearer ${credentials.accessToken}`,
72
+ originator: ORIGINATOR,
73
+ 'openai-alpha': GPT_LIVE_PROTOCOL,
74
+ 'x-session-id': this.sessionIdProvider(),
75
+ };
76
+ if (credentials.accountId) headers['ChatGPT-Account-Id'] = credentials.accountId;
77
+
78
+ const response = await fetch(this.callsUrl, {
79
+ method: 'POST',
80
+ headers,
81
+ body: JSON.stringify({ sdp: input.sdp, session: buildGptLiveSession(input) }),
82
+ signal: AbortSignal.timeout(this.timeoutMs),
83
+ });
84
+ const body = await response.text();
85
+ if (!response.ok) {
86
+ throw new Error(
87
+ `GPT-Live returned ${response.status}: ${body.slice(0, MAX_ERROR_CHARS) || response.statusText}`,
88
+ );
89
+ }
90
+ assertSdp(body, 'answer');
91
+ return { sdp: body, callId: parseCallId(response.headers.get('location')) };
92
+ }
93
+ }
94
+
95
+ function buildGptLiveSession(input: GptLiveCallStart): Record<string, unknown> {
96
+ const session: Record<string, unknown> = {
97
+ instructions: input.instructions,
98
+ model: GPT_LIVE_MODEL,
99
+ audio: { output: { voice: input.voice ?? GPT_LIVE_DEFAULT_VOICE } },
100
+ // `client` delegation keeps every task request on this side, where moxxy
101
+ // runs the user's own words as an agent turn rather than letting a backend
102
+ // model act on its paraphrase.
103
+ delegation: { type: 'client' },
104
+ };
105
+ if (input.history.length > 0) {
106
+ session.initial_items = input.history.map((item) => ({
107
+ type: 'message',
108
+ role: item.role,
109
+ content: [{ type: item.role === 'user' ? 'input_text' : 'output_text', text: item.text }],
110
+ }));
111
+ }
112
+ return session;
113
+ }
114
+
115
+ /**
116
+ * The call carries a live bearer token, so a configurable base URL may only
117
+ * point at https://chatgpt.com or a loopback host (the local test seam).
118
+ */
119
+ export function buildGptLiveCallsUrl(baseUrl = DEFAULT_GPT_LIVE_BASE_URL): string {
120
+ const url = new URL(baseUrl);
121
+ const httpsChatgpt = url.protocol === 'https:' && url.hostname === 'chatgpt.com';
122
+ if (!httpsChatgpt && !isLoopbackHostname(url.hostname)) {
123
+ throw new Error(
124
+ `Refusing to send ChatGPT OAuth credentials to ${url.origin}; GPT-Live must use https://chatgpt.com.`,
125
+ );
126
+ }
127
+ return `${url.origin}${url.pathname.replace(/\/+$/, '')}${CALLS_PATH}${CALLS_QUERY}`;
128
+ }
129
+
130
+ function isLoopbackHostname(hostname: string): boolean {
131
+ const host = hostname.replace(/^\[|\]$/g, '').toLowerCase();
132
+ return host === 'localhost' || host === '127.0.0.1' || host === '::1';
133
+ }
134
+
135
+ function assertSdp(value: string, kind: 'offer' | 'answer'): void {
136
+ if (!value.startsWith('v=0') || Buffer.byteLength(value, 'utf8') > MAX_SDP_BYTES) {
137
+ throw new Error(`Invalid GPT-Live ${kind} SDP`);
138
+ }
139
+ }
140
+
141
+ function parseCallId(location: string | null): string {
142
+ const path = location ? location.split('?', 1).join('') : '';
143
+ const segment = path.split('/').filter(Boolean).at(-1);
144
+ if (!segment || !CALL_ID.test(segment)) {
145
+ throw new Error('GPT-Live answer is missing a valid Location header');
146
+ }
147
+ return segment;
148
+ }
@@ -0,0 +1,90 @@
1
+ import { describe, expect, it } from 'vitest';
2
+ import {
3
+ asEventId,
4
+ asSessionId,
5
+ asTurnId,
6
+ type MoxxyEvent,
7
+ } from '@moxxy/sdk';
8
+ import {
9
+ GPT_LIVE_HISTORY_MAX_ITEMS,
10
+ GPT_LIVE_HISTORY_MAX_ITEM_CHARS,
11
+ buildGptLiveHistory,
12
+ } from './gpt-live-history.js';
13
+
14
+ let seq = 0;
15
+ function base(turn: string) {
16
+ seq += 1;
17
+ return {
18
+ id: asEventId(`event-${seq}`),
19
+ seq,
20
+ ts: seq,
21
+ sessionId: asSessionId('session-1'),
22
+ turnId: asTurnId(turn),
23
+ };
24
+ }
25
+
26
+ function prompt(text: string, turn = `turn-${seq}`): MoxxyEvent {
27
+ return { ...base(turn), type: 'user_prompt', source: 'user', text };
28
+ }
29
+
30
+ function reply(content: string, turn = `turn-${seq}`): MoxxyEvent {
31
+ return { ...base(turn), type: 'assistant_message', source: 'model', content, stopReason: 'end_turn' };
32
+ }
33
+
34
+ describe('buildGptLiveHistory', () => {
35
+ it('maps the chat conversation to ordered user and assistant items', () => {
36
+ const history = buildGptLiveHistory([
37
+ prompt('Sekretne słowo to pomarańcza.'),
38
+ { ...base('turn-x'), type: 'tool_result', source: 'tool', callId: 'c1', output: 'ignored' } as unknown as MoxxyEvent,
39
+ reply('Zapamiętałem sekretne słowo.'),
40
+ ]);
41
+
42
+ expect(history).toEqual([
43
+ { role: 'user', text: 'Sekretne słowo to pomarańcza.' },
44
+ { role: 'assistant', text: 'Zapamiętałem sekretne słowo.' },
45
+ ]);
46
+ });
47
+
48
+ it('skips blank messages and mid-turn checkpoint prompts', () => {
49
+ const checkpoint = {
50
+ ...prompt('Verify your work before finishing.'),
51
+ origin: { kind: 'checkpoint', name: 'verify' },
52
+ } as MoxxyEvent;
53
+
54
+ const history = buildGptLiveHistory([prompt(' '), checkpoint, reply('Gotowe.')]);
55
+
56
+ expect(history).toEqual([{ role: 'assistant', text: 'Gotowe.' }]);
57
+ });
58
+
59
+ it('keeps only the newest items when the conversation exceeds the item limit', () => {
60
+ const events = Array.from({ length: GPT_LIVE_HISTORY_MAX_ITEMS + 10 }, (_, index) =>
61
+ prompt(`wiadomość ${index}`),
62
+ );
63
+
64
+ const history = buildGptLiveHistory(events);
65
+
66
+ expect(history).toHaveLength(GPT_LIVE_HISTORY_MAX_ITEMS);
67
+ expect(history.at(-1)).toEqual({ role: 'user', text: `wiadomość ${GPT_LIVE_HISTORY_MAX_ITEMS + 9}` });
68
+ expect(history[0]).toEqual({ role: 'user', text: 'wiadomość 10' });
69
+ });
70
+
71
+ it('keeps the newest messages that fit the token budget', () => {
72
+ const long = 'x'.repeat(GPT_LIVE_HISTORY_MAX_ITEM_CHARS);
73
+ const events = Array.from({ length: 40 }, (_, index) => prompt(`${index}:${long}`));
74
+
75
+ const history = buildGptLiveHistory(events);
76
+ const approxTokens = history.reduce((sum, item) => sum + Math.ceil(item.text.length / 4), 0);
77
+
78
+ expect(history.length).toBeLessThan(40);
79
+ expect(approxTokens).toBeLessThanOrEqual(8_192);
80
+ expect(history.at(-1)?.text.startsWith('39:')).toBe(true);
81
+ });
82
+
83
+ it('truncates a single oversized message instead of dropping it', () => {
84
+ const history = buildGptLiveHistory([reply('y'.repeat(GPT_LIVE_HISTORY_MAX_ITEM_CHARS * 3))]);
85
+
86
+ expect(history).toHaveLength(1);
87
+ expect(history[0]?.text.length).toBeLessThanOrEqual(GPT_LIVE_HISTORY_MAX_ITEM_CHARS);
88
+ expect(history[0]?.text.endsWith('…')).toBe(true);
89
+ });
90
+ });
@@ -0,0 +1,59 @@
1
+ import type { MoxxyEvent } from '@moxxy/sdk';
2
+
3
+ /** GPT-Live accepts at most 128 initial items totalling 8,192 tokens. */
4
+ export const GPT_LIVE_HISTORY_MAX_ITEMS = 128;
5
+ export const GPT_LIVE_HISTORY_MAX_TOKENS = 8_192;
6
+ /** One message may use at most ~2,048 tokens (a quarter of the budget), so a
7
+ * single long reply cannot crowd every other message out of the model's view. */
8
+ export const GPT_LIVE_HISTORY_MAX_ITEM_CHARS = 8_192;
9
+
10
+ export interface GptLiveHistoryItem {
11
+ readonly role: 'user' | 'assistant';
12
+ readonly text: string;
13
+ }
14
+
15
+ /**
16
+ * Project a moxxy session log onto the conversation GPT-Live sees when a voice
17
+ * call starts: the user prompts and assistant replies, newest kept first until
18
+ * the item or token budget is spent, returned oldest-first.
19
+ */
20
+ export function buildGptLiveHistory(
21
+ events: ReadonlyArray<MoxxyEvent>,
22
+ ): GptLiveHistoryItem[] {
23
+ const newestFirst: GptLiveHistoryItem[] = [];
24
+ let tokens = 0;
25
+ for (let index = events.length - 1; index >= 0; index -= 1) {
26
+ if (newestFirst.length >= GPT_LIVE_HISTORY_MAX_ITEMS) break;
27
+ const item = toHistoryItem(events[index]);
28
+ if (!item) continue;
29
+ const cost = approximateTokens(item.text);
30
+ if (tokens + cost > GPT_LIVE_HISTORY_MAX_TOKENS) break;
31
+ tokens += cost;
32
+ newestFirst.push(item);
33
+ }
34
+ return newestFirst.reverse();
35
+ }
36
+
37
+ function toHistoryItem(event: MoxxyEvent | undefined): GptLiveHistoryItem | null {
38
+ if (event?.type === 'user_prompt' && event.origin?.kind !== 'checkpoint') {
39
+ return fromText('user', event.text);
40
+ }
41
+ if (event?.type === 'assistant_message') return fromText('assistant', event.content);
42
+ return null;
43
+ }
44
+
45
+ function fromText(role: GptLiveHistoryItem['role'], raw: string): GptLiveHistoryItem | null {
46
+ const text = raw.trim();
47
+ if (!text) return null;
48
+ return { role, text: truncate(text) };
49
+ }
50
+
51
+ function truncate(text: string): string {
52
+ if (text.length <= GPT_LIVE_HISTORY_MAX_ITEM_CHARS) return text;
53
+ return `${text.slice(0, GPT_LIVE_HISTORY_MAX_ITEM_CHARS - 1)}…`;
54
+ }
55
+
56
+ /** Same rough four-characters-per-token estimate the Codex client budgets with. */
57
+ function approximateTokens(text: string): number {
58
+ return Math.ceil(text.length / 4);
59
+ }
@@ -10,3 +10,30 @@ it('gives Astra an OAuth-specific operational budget without changing the defaul
10
10
  });
11
11
  expect(DEFAULT_CODEX_MODEL).toBe('gpt-5.6-sol');
12
12
  });
13
+
14
+ it('lists only the models the ChatGPT plan still serves: GPT-6 and GPT-5.6', () => {
15
+ expect(codexModels.map((m) => m.id)).toEqual([
16
+ 'gpt-6-astra',
17
+ 'gpt-6-sol',
18
+ 'gpt-6-luna',
19
+ 'gpt-5.6-sol',
20
+ 'gpt-5.6-terra',
21
+ 'gpt-5.6-luna',
22
+ ]);
23
+ });
24
+
25
+ it("budgets every model to the Codex backend's 272k window less its 5% margin", () => {
26
+ for (const model of codexModels) {
27
+ expect(model).toMatchObject({
28
+ contextWindow: 258_400,
29
+ maxOutputTokens: 16_384,
30
+ supportsTools: true,
31
+ supportsImages: true,
32
+ supportsReasoning: true,
33
+ });
34
+ }
35
+ });
36
+
37
+ it('offers fast mode on every model the plan serves', () => {
38
+ for (const model of codexModels) expect(model.supportsFast, model.id).toBe(true);
39
+ });
package/src/models.ts CHANGED
@@ -9,28 +9,25 @@ import type { ModelDescriptor } from '@moxxy/sdk';
9
9
  // Every Codex-served model is a gpt-5-family reasoning model, so all advertise
10
10
  // `supportsReasoning` — the request already sends `reasoning.summary: 'auto'`;
11
11
  // the per-provider toggle decides whether the summary is surfaced.
12
+ // The Codex backend serves every model below with a 272k window and uses 95%
13
+ // of it (`effective_context_window_percent`), even where the raw API window is
14
+ // 1.05M. Advertising more made the proactive compactor's
15
+ // `estimatedTokens > 0.75 * contextWindow` gate unreachable, so every overflow
16
+ // fell through to the reactive compact-on-overflow retry.
17
+ const CODEX_CONTEXT_WINDOW = 258_400;
18
+ // The OAuth backend rejects `max_output_tokens`; this is only the compactor's
19
+ // reserve for the reply.
20
+ const CODEX_OUTPUT_RESERVE = 16_384;
21
+
12
22
  export const codexModels: ReadonlyArray<ModelDescriptor> = [
13
- // Astra's operational OAuth window: 272k less the agreed 5% safety margin.
14
- // Output is a compaction reserve; the OAuth backend rejects max_output_tokens.
15
- { id: 'gpt-6-astra', contextWindow: 258_400, maxOutputTokens: 16_384, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, hostedTools: ['web_search'] },
16
- // The ChatGPT-plan Codex backend enforces a ~400k window for the gpt-5-family
17
- // models it serves, well below the raw API ceiling. Advertising 1M here made
18
- // the proactive compactor's `estimatedTokens > 0.75 * contextWindow` gate
19
- // unreachable, so every overflow fell through to the reactive
20
- // compact-on-overflow retry. Keep these in step with the rest of the catalog.
21
- // GPT-5.6 family (GA July 9, 2026): served to ChatGPT-Pro/Plus subscribers
22
- // under the same ids the API uses (sol/terra/luna — no `-codex` variant).
23
- // Sol is OpenAI's default Codex model. The ChatGPT-plan window cap applies
24
- // here too, so keep them at 400k like the rest of this list.
25
- { id: 'gpt-5.6-sol', contextWindow: 400_000, maxOutputTokens: 128_000, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, hostedTools: ['web_search'] },
26
- { id: 'gpt-5.6-terra', contextWindow: 400_000, maxOutputTokens: 128_000, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, hostedTools: ['web_search'] },
27
- { id: 'gpt-5.6-luna', contextWindow: 400_000, maxOutputTokens: 128_000, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, hostedTools: ['web_search'] },
28
- { id: 'gpt-5.5', contextWindow: 400_000, maxOutputTokens: 128_000, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, hostedTools: ['web_search'] },
29
- { id: 'gpt-5.4', contextWindow: 400_000, maxOutputTokens: 128_000, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, hostedTools: ['web_search'] },
30
- { id: 'gpt-5.4-mini', contextWindow: 400_000, maxOutputTokens: 128_000, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, hostedTools: ['web_search'] },
31
- { id: 'gpt-5.3-codex', contextWindow: 400_000, maxOutputTokens: 128_000, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, hostedTools: ['web_search'] },
32
- { id: 'gpt-5.3-codex-spark', contextWindow: 400_000, maxOutputTokens: 128_000, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, hostedTools: ['web_search'] },
33
- { id: 'gpt-5.2', contextWindow: 400_000, maxOutputTokens: 128_000, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, hostedTools: ['web_search'] },
23
+ // GPT-6: Astra, Sol (flagship) and Luna (fast).
24
+ { id: 'gpt-6-astra', contextWindow: CODEX_CONTEXT_WINDOW, maxOutputTokens: CODEX_OUTPUT_RESERVE, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, supportsFast: true, hostedTools: ['web_search'] },
25
+ { id: 'gpt-6-sol', contextWindow: CODEX_CONTEXT_WINDOW, maxOutputTokens: CODEX_OUTPUT_RESERVE, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, supportsFast: true, hostedTools: ['web_search'] },
26
+ { id: 'gpt-6-luna', contextWindow: CODEX_CONTEXT_WINDOW, maxOutputTokens: CODEX_OUTPUT_RESERVE, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, supportsFast: true, hostedTools: ['web_search'] },
27
+ // GPT-5.6 family (GA July 9, 2026), under the same ids the API uses.
28
+ { id: 'gpt-5.6-sol', contextWindow: CODEX_CONTEXT_WINDOW, maxOutputTokens: CODEX_OUTPUT_RESERVE, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, supportsFast: true, hostedTools: ['web_search'] },
29
+ { id: 'gpt-5.6-terra', contextWindow: CODEX_CONTEXT_WINDOW, maxOutputTokens: CODEX_OUTPUT_RESERVE, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, supportsFast: true, hostedTools: ['web_search'] },
30
+ { id: 'gpt-5.6-luna', contextWindow: CODEX_CONTEXT_WINDOW, maxOutputTokens: CODEX_OUTPUT_RESERVE, supportsTools: true, supportsStreaming: true, supportsImages: true, supportsDocuments: true, supportsReasoning: true, supportsFast: true, hostedTools: ['web_search'] },
34
31
  ];
35
32
 
36
33
  // OpenAI's own Codex default moved to gpt-5.6-sol at GA (July 9, 2026); mirror
@@ -7,6 +7,7 @@ import { CODEX_RESPONSES_URL } from './oauth.js';
7
7
  import type { CodexTokens } from './types.js';
8
8
  import type { ProviderEvent, ProviderRequest } from '@moxxy/sdk';
9
9
  import { assertDefined, defineTool, z } from '@moxxy/sdk';
10
+ import { removeDir } from '@moxxy/vitest-preset/fs';
10
11
 
11
12
  // The refresh path takes a cross-process lockfile under `<moxxy home>/locks`;
12
13
  // point MOXXY_HOME at a temp dir so tests never touch the real ~/.moxxy.
@@ -19,7 +20,7 @@ beforeAll(async () => {
19
20
  afterAll(async () => {
20
21
  if (priorMoxxyHome === undefined) delete process.env.MOXXY_HOME;
21
22
  else process.env.MOXXY_HOME = priorMoxxyHome;
22
- await fs.rm(moxxyHomeTmp, { recursive: true, force: true });
23
+ await removeDir(moxxyHomeTmp);
23
24
  });
24
25
 
25
26
  function makeTokens(overrides: Partial<CodexTokens> = {}): CodexTokens {
@@ -50,7 +51,7 @@ async function collect<T>(it: AsyncIterable<T>): Promise<T[]> {
50
51
 
51
52
  function baseRequest(over: Partial<ProviderRequest> = {}): ProviderRequest {
52
53
  return {
53
- model: 'gpt-5.3-codex',
54
+ model: 'gpt-5.6-sol',
54
55
  messages: [{ role: 'user', content: [{ type: 'text', text: 'hi' }] }],
55
56
  ...over,
56
57
  };
@@ -124,7 +125,7 @@ describe('CodexProvider.stream', () => {
124
125
  expect(h['Accept']).toBe('text/event-stream');
125
126
 
126
127
  // Event sequence: message_start, text_delta('hello'), message_end (with usage)
127
- expect(events[0]).toMatchObject({ type: 'message_start', model: 'gpt-5.3-codex' });
128
+ expect(events[0]).toMatchObject({ type: 'message_start', model: 'gpt-5.6-sol' });
128
129
  expect(events.some((e) => e.type === 'text_delta' && e.delta === 'hello')).toBe(true);
129
130
  const end = events.find((e): e is Extract<ProviderEvent, { type: 'message_end' }> => e.type === 'message_end');
130
131
  expect(end?.usage).toEqual({ inputTokens: 3, outputTokens: 5 });
@@ -211,6 +212,19 @@ describe('CodexProvider.stream', () => {
211
212
  expect(bodies[1]?.reasoning).toMatchObject({ effort: 'high' });
212
213
  });
213
214
 
215
+ it('sends the xhigh reasoning effort a request asks for', async () => {
216
+ let body: Record<string, unknown> = {};
217
+ const fakeFetch = vi.fn(async (_u: RequestInfo | URL, init?: RequestInit) => {
218
+ body = JSON.parse(String(init?.body)) as Record<string, unknown>;
219
+ return new Response(sseStream(['data: {"type":"response.completed"}\n\n']), { status: 200 });
220
+ });
221
+ const provider = new CodexProvider({ tokens: makeTokens(), fetch: fakeFetch as unknown as typeof fetch });
222
+
223
+ await collect(provider.stream({ ...baseRequest(), reasoning: { effort: 'xhigh' } }));
224
+
225
+ expect(body.reasoning).toMatchObject({ effort: 'xhigh' });
226
+ });
227
+
214
228
  it('uses a stable default session id across turns so the prefix cache can hit', async () => {
215
229
  const keys: unknown[] = [];
216
230
  const fakeFetch = vi.fn(async (_u: RequestInfo | URL, init?: RequestInit) => {
@@ -465,13 +479,13 @@ describe('CodexProvider.stream', () => {
465
479
  const provider = new CodexProvider({ tokens: makeTokens() });
466
480
  const bigImage = 'A'.repeat(4 * 1024 * 1024); // 4 MiB of base64
467
481
  const withImage = await provider.countTokens({
468
- model: 'gpt-5.3-codex',
482
+ model: 'gpt-5.6-sol',
469
483
  messages: [
470
484
  { role: 'user', content: [{ type: 'image', mediaType: 'image/png', data: bigImage }] },
471
485
  ],
472
486
  });
473
487
  const textOnly = await provider.countTokens({
474
- model: 'gpt-5.3-codex',
488
+ model: 'gpt-5.6-sol',
475
489
  messages: [{ role: 'user', content: [{ type: 'text', text: 'hi' }] }],
476
490
  });
477
491
  // A 4 MiB base64 image must NOT be counted as ~1M tokens (4MiB/4) — the
package/src/provider.ts CHANGED
@@ -62,7 +62,7 @@ export interface CodexProviderConfig {
62
62
  * moxxy.config.ts (e.g. `{ reasoningEffort: 'high' }`); the CLI's
63
63
  * credential resolution merges that config through to `createClient`.
64
64
  */
65
- readonly reasoningEffort?: 'low' | 'medium' | 'high';
65
+ readonly reasoningEffort?: 'low' | 'medium' | 'high' | 'xhigh';
66
66
  /** Test seam — when omitted we use the global `fetch`. */
67
67
  readonly fetch?: typeof fetch;
68
68
  /** Test seam — when omitted we use crypto.randomUUID for the per-request session id. */
@@ -96,7 +96,7 @@ export class CodexProvider implements LLMProvider {
96
96
  private readonly onTokensRefreshed?: (next: CodexTokens) => void | Promise<void>;
97
97
  private readonly reloadTokens?: () => Promise<CodexTokens | null>;
98
98
  private readonly defaultModel: string;
99
- private readonly reasoningEffort?: 'low' | 'medium' | 'high';
99
+ private readonly reasoningEffort?: 'low' | 'medium' | 'high' | 'xhigh';
100
100
  private readonly fetchImpl: typeof fetch;
101
101
  private readonly sessionIdProvider: () => string;
102
102
  private readonly idleTimeoutMs: number;
@@ -141,7 +141,7 @@ describe('extractSystemText', () => {
141
141
 
142
142
  describe('toResponsesBody', () => {
143
143
  const req = {
144
- model: 'gpt-5.3-codex',
144
+ model: 'gpt-5.6-sol',
145
145
  messages: [
146
146
  { role: 'system' as const, content: [{ type: 'text' as const, text: 'BASE' }] },
147
147
  { role: 'user' as const, content: [{ type: 'text' as const, text: 'hi' }] },
@@ -175,4 +175,9 @@ describe('toResponsesBody', () => {
175
175
  { type: 'web_search' },
176
176
  ]);
177
177
  });
178
+
179
+ it('asks for the fast tier only when fast mode is on', () => {
180
+ expect(toResponsesBody({ ...req, fast: true }).service_tier).toBe('priority');
181
+ expect(toResponsesBody(req)).not.toHaveProperty('service_tier');
182
+ });
178
183
  });