llm-switcher 1.1.6 → 1.1.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -191,7 +191,11 @@ flowchart LR
191
191
 
192
192
  ---
193
193
 
194
- ## Changes in 1.1.6
194
+ ## Changes in 1.1.7
195
+
196
+ - **Codex tools.** With `publicModels` set, Codex lost its tools and ended after one answer. The model catalog copied the metadata of a real OpenAI model, which puts Codex in the "Responses Lite" form. The catalog now keeps Codex in its direct tool mode, and the gateway also reads tools that arrive as an `additional_tools` input item.
197
+
198
+ ### Changes in 1.1.6
195
199
 
196
200
  - **Codex over WebSocket.** The gateway keeps the turns of each WebSocket session. A turn that sends `previous_response_id` gets the earlier turns back, so Codex no longer loses the task after the first tool call. An unknown id fails the turn with `previous_response_not_found`.
197
201
  - **Codex warmup.** A `response.create` frame with `generate: false` gets a local answer. It no longer spends a model call.
package/README.vi.md CHANGED
@@ -191,7 +191,11 @@ flowchart LR
191
191
 
192
192
  ---
193
193
 
194
- ## Thay đổi trong bản 1.1.6
194
+ ## Thay đổi trong bản 1.1.7
195
+
196
+ - **Tool của Codex.** Khi có `publicModels`, Codex mất hết tool và dừng sau một câu trả lời. Model catalog chép metadata của một model OpenAI thật, và metadata này đưa Codex sang dạng "Responses Lite". Giờ catalog giữ Codex ở chế độ tool trực tiếp, và gateway cũng đọc tool gửi đến dưới dạng input item `additional_tools`.
197
+
198
+ ### Thay đổi trong bản 1.1.6
195
199
 
196
200
  - **Codex qua WebSocket.** Gateway giữ các lượt của mỗi phiên WebSocket. Lượt nào gửi `previous_response_id` sẽ nhận lại các lượt trước, nên Codex không còn mất nhiệm vụ sau lần gọi tool đầu tiên. Id không tồn tại làm lượt đó lỗi với `previous_response_not_found`.
197
201
  - **Warmup của Codex.** Frame `response.create` có `generate: false` được trả lời ngay tại máy. Frame này không còn tốn một lần gọi model.
package/formats.mjs CHANGED
@@ -612,6 +612,8 @@ function responsesToIR(payload) {
612
612
  else ir.messages.push({ role: 'assistant', toolCalls: [call] });
613
613
  };
614
614
 
615
+ // Responses Lite sends the tools as an input item, not in payload.tools.
616
+ const toolList = Array.isArray(payload.tools) ? [...payload.tools] : [];
615
617
  const input = payload.input;
616
618
  if (typeof input === 'string' && input) {
617
619
  ir.messages.push({ role: 'user', content: input });
@@ -655,14 +657,19 @@ function responsesToIR(payload) {
655
657
  pushCall({ id: item.call_id || item.id || null, name: 'local_shell', args: { command: a.command || [], workdir: a.working_directory ?? undefined, timeout_ms: a.timeout_ms ?? undefined } });
656
658
  } else if (item.type === 'function_call_output' || item.type === 'custom_tool_call_output' || item.type === 'local_shell_call_output') {
657
659
  ir.messages.push({ role: 'tool', toolCallId: item.call_id || item.id, content: responsesOutputToText(item.output) });
660
+ } else if (item.type === 'additional_tools' && Array.isArray(item.tools)) {
661
+ toolList.push(...item.tools);
658
662
  }
659
663
  }
660
664
  }
661
665
 
662
- if (Array.isArray(payload.tools) && payload.tools.length) {
666
+ if (toolList.length) {
663
667
  ir.toolMeta = {};
664
668
  const addTool = (t, namespace) => {
665
669
  if (!t || typeof t !== 'object') return;
670
+ // A tool in both payload.tools and an additional_tools item is sent once: upstreams refuse duplicate names.
671
+ if (t.name && t.type !== 'namespace' && ir.toolMeta[responsesToolName(namespace, t.name)]) return;
672
+ if (t.type === 'local_shell' && ir.toolMeta.local_shell) return;
666
673
  if (t.type === 'namespace' && Array.isArray(t.tools)) {
667
674
  for (const inner of t.tools) addTool(inner, t.name);
668
675
  return;
@@ -681,7 +688,7 @@ function responsesToIR(payload) {
681
688
  }
682
689
  // web_search / file_search / tool_search / image_generation...: hosted tools, skipped.
683
690
  };
684
- for (const t of payload.tools) addTool(t, null);
691
+ for (const t of toolList) addTool(t, null);
685
692
  }
686
693
 
687
694
  const tc = payload.tool_choice;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "llm-switcher",
3
- "version": "1.1.6",
3
+ "version": "1.1.7",
4
4
  "description": "Zero-dependency multi-protocol edge gateway & provider switcher for Claude Code, Codex, OpenAI and Gemini clients",
5
5
  "keywords": [
6
6
  "llm",
package/state.mjs CHANGED
@@ -346,8 +346,12 @@ export function buildCodexCatalog(profile) {
346
346
 
347
347
  // One catalog entry. The catalog file and the gateway's /v1/models both use it, so they cannot drift.
348
348
  export function codexModelEntry(name, is1M) {
349
+ // The template copies a real OpenAI model. Its Responses Lite and code modes send the tools in a
350
+ // form made for OpenAI's own tools, and an upstream then gets none (LS-5).
351
+ const { tool_mode, ...template } = structuredClone(codexCatalogTemplate() || {});
349
352
  return {
350
- ...structuredClone(codexCatalogTemplate() || {}),
353
+ ...template,
354
+ use_responses_lite: false,
351
355
  slug: name,
352
356
  display_name: name,
353
357
  ...(is1M ? { context_window: 1000000, max_context_window: 1000000 } : {})
@@ -792,3 +792,10 @@ test('allowed_tools that matches no declared tool gives a request without tools,
792
792
  }
793
793
  }
794
794
  });
795
+
796
+ test('responsesToIR reads additional_tools items and sends a tool once', () => {
797
+ const tool = { type: 'function', name: 'exec_command', parameters: { type: 'object' } };
798
+ const ir = responsesToIR({ model: 'm', tools: [tool], input: [{ type: 'additional_tools', role: 'developer', tools: [tool, { type: 'custom', name: 'apply_patch' }] }] });
799
+ assert.deepEqual(ir.tools.map(t => t.name), ['exec_command', 'apply_patch']);
800
+ assert.equal(ir.messages.length, 0, 'the item is not a message');
801
+ });
@@ -845,6 +845,26 @@ test('Codex WS: a generate:false warmup is answered locally, never sent upstream
845
845
  ws.socket.destroy();
846
846
  });
847
847
 
848
+ // LS-5: in the Responses Lite form Codex sends its tools as an additional_tools input item of the
849
+ // warmup, not as a tools field. The next turn must reach upstream with those tools.
850
+ test('Codex WS: tools sent as an additional_tools item reach upstream', async () => {
851
+ const ws = await rawWs();
852
+ ws.socket.write(clientFrame(1, JSON.stringify({ type: 'response.create', model: 'main', generate: false, input: [
853
+ { type: 'additional_tools', id: 'at_1', role: 'developer', tools: [
854
+ { type: 'function', name: 'exec_command', description: 'Run a command', parameters: { type: 'object', properties: { cmd: { type: 'string' } } } },
855
+ { type: 'custom', name: 'apply_patch', description: 'Patch files' }] },
856
+ { type: 'message', role: 'developer', content: [{ type: 'input_text', text: 'You are a coding agent.' }] }] })));
857
+ assert.ok(await until(() => ws.messages.some(m => m.type === 'response.completed')), 'the warmup completes');
858
+ const warmId = ws.messages.find(m => m.type === 'response.completed').response.id;
859
+ const before = received.length;
860
+ ws.socket.write(clientFrame(1, JSON.stringify({ type: 'response.create', model: 'main', previous_response_id: warmId,
861
+ input: [{ type: 'message', role: 'user', content: [{ type: 'input_text', text: 'fix calc.py' }] }] })));
862
+ assert.ok(await until(() => received.length > before && ws.messages.filter(m => m.type === 'response.completed' || m.type === 'response.failed').length === 2), 'turn 2 ends');
863
+ const sent = JSON.stringify(received.slice(before).at(-1).body);
864
+ assert.ok(sent.includes('exec_command') && sent.includes('apply_patch'), `the tools are lost: ${sent.slice(0, 400)}`);
865
+ ws.socket.destroy();
866
+ });
867
+
848
868
  test('Codex WS: a mid-stream error is logged with its text', async () => {
849
869
  const ws = await rawWs();
850
870
  ws.socket.write(clientFrame(1, JSON.stringify({ type: 'response.create', model: 'main', input: 'MID_STREAM_ERROR over ws' })));
@@ -699,3 +699,13 @@ test('an empty launch lock is held, not taken over', async (t) => {
699
699
  assert.ok(waited >= 100 && waited < 4000, `the writer waited ${waited} ms`);
700
700
  assert.equal(fs.existsSync(lock), false);
701
701
  });
702
+
703
+ // LS-5: the template copies a real OpenAI model that runs Codex in Responses Lite and code mode.
704
+ // Those modes send the tools in a form made for OpenAI's own tools, so every entry turns them off.
705
+ test('buildCodexCatalog entries keep Codex in its direct tool mode', () => {
706
+ const catalog = buildCodexCatalog({ inFormat: 'responses', publicModels: ['gpt-5.6-sol'] });
707
+ for (const entry of catalog.models) {
708
+ assert.equal(entry.use_responses_lite, false);
709
+ assert.equal(Object.hasOwn(entry, 'tool_mode'), false);
710
+ }
711
+ });