openzoo 0.48.22 → 0.48.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,425 @@
1
+ /**
2
+ * OpenAI Responses API <-> chat completions.
3
+ *
4
+ * WHY THIS EXISTS: some harnesses speak ONLY the Responses API. OpenAI's Codex
5
+ * Security CLI is the case that forced it — its provider table pins
6
+ * `wire_api: "responses"`, so pointing OPENAI_BASE_URL at this proxy got a bare
7
+ * 404 on /v1/responses and the tool fell back to `wss://api.openai.com`,
8
+ * bypassing the proxy entirely while looking like an auth failure.
9
+ *
10
+ * The zoo serves chat completions. Rather than teach the gateway a second wire
11
+ * format, translate at the edge: Responses in, chat out, chat back, Responses
12
+ * back. Everything between (payment, auto-bind, replay cache, metering) stays
13
+ * on one code path and cannot drift.
14
+ *
15
+ * Mirrors the shape already used for Anthropic in ./anthropic.js.
16
+ */
17
+
18
+
19
+ /**
20
+ * Flatten Codex's `additional_tools` into chat-completions tools.
21
+ *
22
+ * Codex does NOT use the `tools` parameter. It declares capability as an input
23
+ * item `{type:"additional_tools", role:"developer", tools:[...]}` whose entries
24
+ * are either NAMESPACES (a named bag of tools) or leaf tools. Leaves come in two
25
+ * flavours:
26
+ *
27
+ * type:"function" — ordinary JSON-schema tool. Maps straight across.
28
+ * type:"custom" — FREEFORM: its input is raw text, not JSON. `exec` is one,
29
+ * and it is the tool through which Codex runs everything
30
+ * (shell included, via tools.exec_command inside the JS).
31
+ *
32
+ * Chat completions has no freeform tool type, so a custom tool is wrapped as a
33
+ * function with a single string property and unwrapped on the way back. Without
34
+ * this the model is handed NO tools at all, answers in prose, never runs a
35
+ * command, and the scan dies with "did not create required draft artifacts" —
36
+ * which is what happened on every repo.
37
+ *
38
+ * `custom` collects the names that need unwrapping, because the response
39
+ * translator has to know which calls to re-emit as custom_tool_call.
40
+ */
41
+ export function toolsFromAdditional(item, custom) {
42
+ const out = [];
43
+ const walk = (list) => {
44
+ for (const t of list || []) {
45
+ if (t?.type === 'namespace') { walk(t.tools); continue; }
46
+ if (!t?.name) continue;
47
+ if (t.type === 'custom') {
48
+ custom.add(t.name);
49
+ out.push({
50
+ type: 'function',
51
+ function: {
52
+ name: t.name,
53
+ description: t.description || '',
54
+ parameters: {
55
+ type: 'object',
56
+ properties: {
57
+ input: { type: 'string', description: 'Raw tool input, verbatim. Not JSON, not fenced.' },
58
+ },
59
+ required: ['input'],
60
+ },
61
+ },
62
+ });
63
+ } else {
64
+ out.push({
65
+ type: 'function',
66
+ function: { name: t.name, description: t.description || '', parameters: t.parameters || { type: 'object', properties: {} } },
67
+ });
68
+ }
69
+ }
70
+ };
71
+ walk(item?.tools);
72
+ return out;
73
+ }
74
+
75
+
76
+ /**
77
+ * Flatten a tool result into text.
78
+ *
79
+ * `output` is NOT a string. Codex sends an array of content parts —
80
+ * [{type:"input_text", text:"..."}, ...] — and `String()` on that yields
81
+ * "[object Object],[object Object]".
82
+ *
83
+ * MEASURED: every one of an agent's 51 tool calls came back as [object Object].
84
+ * It ran real commands with exit_code 0, received nothing legible, and could
85
+ * never learn enough to write its output files. The scan then failed with
86
+ * "did not create required draft artifacts", and the artifact directories were
87
+ * EMPTY rather than partial — which is the tell that the agent was blind, not
88
+ * interrupted.
89
+ */
90
+ function outputText(output) {
91
+ if (output == null) return '';
92
+ if (typeof output === 'string') return output;
93
+ if (Array.isArray(output)) {
94
+ return output.map((part) => {
95
+ if (typeof part === 'string') return part;
96
+ if (part && typeof part === 'object') {
97
+ // text / input_text / output_text all carry `text`; anything else is
98
+ // structured and JSON is the honest rendering of it.
99
+ if (typeof part.text === 'string') return part.text;
100
+ return JSON.stringify(part);
101
+ }
102
+ return String(part);
103
+ }).join('');
104
+ }
105
+ if (typeof output === 'object') {
106
+ if (typeof output.text === 'string') return output.text;
107
+ if (typeof output.content === 'string') return output.content;
108
+ return JSON.stringify(output);
109
+ }
110
+ return String(output);
111
+ }
112
+
113
+ /**
114
+ * Responses request -> chat-completions request.
115
+ *
116
+ * `input` is the polymorphic field: a bare string, a list of role/content
117
+ * turns, or a list of typed content parts. All three are real and a harness
118
+ * will send whichever it feels like, so all three are handled rather than
119
+ * assuming the documented one.
120
+ */
121
+ export function responsesToChat(body, meta = {}) {
122
+ const custom = meta.custom || (meta.custom = new Set());
123
+ const extraTools = [];
124
+ let messages = [];
125
+
126
+ if (typeof body.input === 'string') {
127
+ messages = [{ role: 'user', content: body.input }];
128
+ } else if (Array.isArray(body.input)) {
129
+ messages = body.input.map((part) => {
130
+ if (part && typeof part === 'object') {
131
+ // The agent loop feeds tool RESULTS back as input items. Without this
132
+ // they become user prose, the model loses the call/result pairing and
133
+ // re-issues the same tool call forever.
134
+ if (part.type === 'additional_tools') {
135
+ extraTools.push(...toolsFromAdditional(part, custom));
136
+ return null; // it is a tool DECLARATION, not a turn
137
+ }
138
+ if (part.type === 'custom_tool_call_output') {
139
+ return { role: 'tool', tool_call_id: part.call_id, content: outputText(part.output) };
140
+ }
141
+ if (part.type === 'custom_tool_call') {
142
+ return {
143
+ role: 'assistant', content: null,
144
+ tool_calls: [{
145
+ id: part.call_id, type: 'function',
146
+ function: { name: part.name, arguments: JSON.stringify({ input: part.input ?? '' }) },
147
+ }],
148
+ };
149
+ }
150
+ if (part.type === 'function_call_output') {
151
+ return { role: 'tool', tool_call_id: part.call_id, content: outputText(part.output) };
152
+ }
153
+ if (part.type === 'function_call') {
154
+ return {
155
+ role: 'assistant',
156
+ content: null,
157
+ tool_calls: [{
158
+ id: part.call_id, type: 'function',
159
+ function: { name: part.name, arguments: part.arguments ?? '{}' },
160
+ }],
161
+ };
162
+ }
163
+ if (part.role) {
164
+ // content may itself be an array of {type:'input_text', text}
165
+ const c = part.content;
166
+ if (Array.isArray(c)) {
167
+ const text = c
168
+ .map((seg) => (typeof seg === 'string' ? seg : seg?.text ?? ''))
169
+ .join('');
170
+ return { role: part.role, content: text };
171
+ }
172
+ return { role: part.role, content: c ?? part.text ?? '' };
173
+ }
174
+ return { role: 'user', content: part.text ?? JSON.stringify(part) };
175
+ }
176
+ return { role: 'user', content: String(part) };
177
+ }).filter(Boolean);
178
+ } else if (Array.isArray(body.messages)) {
179
+ messages = body.messages.map((m) => ({ role: m.role || 'user', content: m.content }));
180
+ }
181
+
182
+ // `instructions` is the Responses API's system prompt. Dropping it silently
183
+ // changes the model's behaviour with no error anywhere, which is the worst
184
+ // kind of translation bug.
185
+ if (typeof body.instructions === 'string' && body.instructions.trim()) {
186
+ messages.unshift({ role: 'system', content: body.instructions });
187
+ }
188
+
189
+ const out = {
190
+ model: body.model,
191
+ messages,
192
+ };
193
+
194
+ // DO NOT INVENT A COMPLETION CAP.
195
+ //
196
+ // Codex sends `reasoning: {effort: "xhigh"}` and NO max_output_tokens — it
197
+ // wants the model's natural limit. Defaulting to 4096 here silently truncated
198
+ // every turn of an agent whose reasoning tokens come out of the same budget,
199
+ // so it explored the repo across 51 tool calls and was cut off before it
200
+ // could write scan-manifest.json / findings.json / coverage.json. The scan
201
+ // then failed with "did not create required draft artifacts", which reads as
202
+ // a filesystem problem and is not one.
203
+ //
204
+ // Only pass a cap the caller actually asked for.
205
+ const cap = body.max_output_tokens ?? body.max_tokens;
206
+ if (cap != null) out.max_tokens = cap;
207
+ if (body.temperature != null) out.temperature = body.temperature;
208
+ if (body.top_p != null) out.top_p = body.top_p;
209
+ if (body.stream) out.stream = true;
210
+ // TOOL SHAPES DIFFER. Responses is FLAT ({type,name,parameters}); chat
211
+ // completions is NESTED ({type,function:{name,parameters}}). Forwarding the
212
+ // flat shape unchanged makes the upstream ignore the tools entirely — the
213
+ // model then answers in prose, the agent never runs a shell command, and the
214
+ // scan fails with "did not create required draft artifacts". Nothing errors.
215
+ if (Array.isArray(body.tools)) {
216
+ out.tools = body.tools.map((t) => (t && t.type === 'function' && !t.function
217
+ ? { type: 'function', function: { name: t.name, description: t.description, parameters: t.parameters } }
218
+ : t));
219
+ }
220
+ if (extraTools.length) out.tools = [...(out.tools || []), ...extraTools];
221
+ if (body.tool_choice) out.tool_choice = body.tool_choice;
222
+ return out;
223
+ }
224
+
225
+ /**
226
+ * chat-completions response -> Responses response.
227
+ *
228
+ * `output_text` is included because most SDK readers reach for it first; the
229
+ * structured `output` array is what stricter clients parse. Emitting only one
230
+ * of them makes the response work in some clients and silently look empty in
231
+ * others.
232
+ */
233
+ export function chatToResponses(data, requestedModel, custom) {
234
+ const choice = (data?.choices || [])[0] || {};
235
+ const msg = choice.message || {};
236
+ const text = typeof msg.content === 'string' ? msg.content : '';
237
+ const u = data?.usage || {};
238
+
239
+ const finish = choice.finish_reason;
240
+ // Responses uses status, not finish_reason. "incomplete" is its word for a
241
+ // length cut-off; mapping that to "completed" would tell a caller a truncated
242
+ // answer was whole.
243
+ // `tool_calls` is a NORMAL completion — the model finished its turn by asking
244
+ // for a tool. Marking it incomplete makes the agent treat a healthy turn as a
245
+ // truncation and abandon the loop.
246
+ const status =
247
+ finish === 'length' ? 'incomplete'
248
+ : finish === 'content_filter' ? 'incomplete'
249
+ : 'completed';
250
+
251
+ const out = {
252
+ id: data?.id || `resp_${Math.random().toString(36).slice(2)}`,
253
+ object: 'response',
254
+ created_at: data?.created || Math.floor(Date.now() / 1000),
255
+ status,
256
+ model: data?.model || requestedModel,
257
+ output: buildOutput(data, msg, text, custom),
258
+ output_text: text,
259
+ usage: {
260
+ input_tokens: u.prompt_tokens ?? 0,
261
+ output_tokens: u.completion_tokens ?? 0,
262
+ total_tokens: u.total_tokens ?? 0,
263
+ },
264
+ };
265
+ if (status === 'incomplete') {
266
+ out.incomplete_details = { reason: finish === 'length' ? 'max_output_tokens' : 'content_filter' };
267
+ }
268
+ if (msg.refusal) out.output[0].content[0].refusal = msg.refusal;
269
+ // Keep the payment receipt visible on this shape too — the whole product is
270
+ // that you can see what a call cost.
271
+ if (data?.x402) out.x402 = data.x402;
272
+ return out;
273
+ }
274
+
275
+ /**
276
+ * The `output` array: text and/or tool calls.
277
+ *
278
+ * A tool call is NOT a message with a funny payload — Responses models it as a
279
+ * separate `function_call` item with its own `call_id`, and that id is what the
280
+ * agent echoes back in `function_call_output`. Emitting a text message instead
281
+ * loses the id and the loop cannot close.
282
+ */
283
+ function buildOutput(data, msg, text, custom) {
284
+ const items = [];
285
+ if (text) {
286
+ items.push({
287
+ id: `msg_${(data?.id || '').slice(-16) || 'x'}`,
288
+ type: 'message',
289
+ role: 'assistant',
290
+ status: 'completed',
291
+ content: [{ type: 'output_text', text, annotations: msg.annotations || [] }],
292
+ });
293
+ }
294
+ for (const tc of msg.tool_calls || []) {
295
+ const name = tc.function?.name;
296
+ const rawArgs = tc.function?.arguments ?? '{}';
297
+ // A CUSTOM tool must go back as custom_tool_call carrying RAW text. We
298
+ // wrapped its input in {"input": "..."} on the way out so a JSON-only chat
299
+ // API could carry it; unwrap it here or Codex receives a JSON blob where it
300
+ // expects JavaScript source and the exec sandbox rejects every call.
301
+ if (custom?.has(name)) {
302
+ let input = rawArgs;
303
+ try {
304
+ const parsed = JSON.parse(rawArgs);
305
+ if (parsed && typeof parsed.input === 'string') input = parsed.input;
306
+ } catch { /* model emitted bare text; pass it through */ }
307
+ items.push({
308
+ id: `ctc_${tc.id}`, type: 'custom_tool_call', status: 'completed',
309
+ call_id: tc.id, name, input,
310
+ });
311
+ continue;
312
+ }
313
+ items.push({
314
+ id: `fc_${tc.id}`,
315
+ type: 'function_call',
316
+ status: 'completed',
317
+ call_id: tc.id,
318
+ name,
319
+ arguments: rawArgs,
320
+ });
321
+ }
322
+ return items;
323
+ }
324
+
325
+ /**
326
+ * Emit a chat completion as a Responses-API SSE stream.
327
+ *
328
+ * WHY: a Responses client does not read a JSON body — it opens an event stream
329
+ * and waits for `response.completed`. Returning the object alone closes the
330
+ * socket with the stream half-told, which the client reports as
331
+ * "stream disconnected before completion: stream closed before
332
+ * response.completed" and retries forever. MEASURED against codex-security:
333
+ * every repo failed this way on the first pass.
334
+ *
335
+ * The event NAMES are the contract, not the payload shape. A client that sees
336
+ * an unknown event ignores it; a client that never sees `response.completed`
337
+ * hangs. So the terminal events matter far more than the deltas.
338
+ */
339
+ export function writeResponsesSse(res, data, requestedModel, upstream, custom) {
340
+ const full = chatToResponses(data, requestedModel, custom);
341
+ const text = full.output_text || '';
342
+ const msgItem = full.output.find((o) => o.type === 'message');
343
+ const calls = full.output.filter((o) => o.type === 'function_call' || o.type === 'custom_tool_call');
344
+ const itemId = msgItem?.id || `msg_${Math.random().toString(36).slice(2, 10)}`;
345
+
346
+ const h = {
347
+ 'content-type': 'text/event-stream',
348
+ 'cache-control': 'no-cache',
349
+ connection: 'keep-alive',
350
+ // Without this a proxy in front of us buffers the whole stream and the
351
+ // client sees nothing until the end — which looks exactly like a hang.
352
+ 'x-accel-buffering': 'no',
353
+ };
354
+ const settle = upstream?.headers?.get?.('x-payment-response');
355
+ if (settle) h['x-payment-response'] = settle;
356
+ res.writeHead(200, h);
357
+
358
+ let seq = 0;
359
+ const send = (type, payload) => {
360
+ res.write(`event: ${type}\n`);
361
+ res.write(`data: ${JSON.stringify({ type, sequence_number: seq++, ...payload })}\n\n`);
362
+ };
363
+
364
+ // in_progress shell first — the client builds its item tree from these
365
+ const shell = { ...full, status: 'in_progress', output: [], output_text: '' };
366
+ send('response.created', { response: shell });
367
+ send('response.in_progress', { response: shell });
368
+
369
+ const item = {
370
+ id: itemId, type: 'message', role: 'assistant', status: 'in_progress', content: [],
371
+ };
372
+ send('response.output_item.added', { output_index: 0, item });
373
+ send('response.content_part.added', {
374
+ item_id: itemId, output_index: 0, content_index: 0,
375
+ part: { type: 'output_text', text: '', annotations: [] },
376
+ });
377
+
378
+ // One delta carrying the whole text. The upstream already finished, so
379
+ // chunking it would fake a progressive generation that did not happen.
380
+ if (text) {
381
+ send('response.output_text.delta', {
382
+ item_id: itemId, output_index: 0, content_index: 0, delta: text,
383
+ });
384
+ }
385
+ send('response.output_text.done', {
386
+ item_id: itemId, output_index: 0, content_index: 0, text,
387
+ });
388
+ send('response.content_part.done', {
389
+ item_id: itemId, output_index: 0, content_index: 0,
390
+ part: { type: 'output_text', text, annotations: [] },
391
+ });
392
+ send('response.output_item.done', {
393
+ output_index: 0, item: { ...item, status: 'completed', content: [{ type: 'output_text', text, annotations: [] }] },
394
+ });
395
+
396
+ // TOOL CALLS GET THEIR OWN ITEMS. The non-streaming path already returns
397
+ // them; omitting them here would make the agent work when it does not stream
398
+ // and silently do nothing when it does — which is the configuration it
399
+ // actually uses.
400
+ calls.forEach((call, i) => {
401
+ const idx = i + 1;
402
+ const shellItem = {
403
+ id: call.id, type: 'function_call', status: 'in_progress',
404
+ call_id: call.call_id, name: call.name, arguments: '',
405
+ };
406
+ // A custom tool streams under DIFFERENT event names. Sending
407
+ // function_call_arguments.* for one leaves the client with an item it never
408
+ // sees populated.
409
+ const isCustom = call.type === 'custom_tool_call';
410
+ const payload = isCustom ? call.input : call.arguments;
411
+ send('response.output_item.added', { output_index: idx, item: { ...shellItem, type: call.type } });
412
+ send(isCustom ? 'response.custom_tool_call_input.delta' : 'response.function_call_arguments.delta', {
413
+ item_id: call.id, output_index: idx, delta: payload,
414
+ });
415
+ send(isCustom ? 'response.custom_tool_call_input.done' : 'response.function_call_arguments.done', {
416
+ item_id: call.id, output_index: idx, ...(isCustom ? { input: payload } : { arguments: payload }),
417
+ });
418
+ send('response.output_item.done', { output_index: idx, item: { ...call, status: 'completed' } });
419
+ });
420
+
421
+ // The one event the client is actually waiting for.
422
+ send(full.status === 'completed' ? 'response.completed' : 'response.incomplete', { response: full });
423
+ res.write('data: [DONE]\n\n');
424
+ res.end();
425
+ }
package/lib/tunnel.js CHANGED
@@ -71,7 +71,14 @@ export async function ensureCloudflared(log = console.log) {
71
71
  fs.mkdirSync(BIN_DIR, { recursive: true, mode: 0o700 });
72
72
  const url = `https://github.com/cloudflare/cloudflared/releases/latest/download/${asset.name}`;
73
73
  log(`fetching cloudflared for ${process.platform}/${process.arch} (one-time, ~35MB)...`);
74
- const r = await fetch(url, { redirect: 'follow' });
74
+ // BOUND IT. An unbounded fetch on a filtered/slow network never settles, and
75
+ // the caller is a background task whose failure the user cannot see — so the
76
+ // public URL simply never appeared and nothing said why. 120s is generous for
77
+ // 35MB and still finite; on timeout the tunnel degrades to local-only, which
78
+ // is the documented fallback, instead of hanging out of sight forever.
79
+ const ms = Number(process.env.OPENZOO_CF_TIMEOUT_MS || 120000);
80
+ const r = await fetch(url, { redirect: 'follow', signal: AbortSignal.timeout(ms) })
81
+ .catch((e) => { throw new Error(`cloudflared download failed (${e.name === 'TimeoutError' ? `no response in ${ms / 1000}s` : e.message}) — set OPENZOO_NO_TUNNEL=1 to skip the public URL`); });
75
82
  if (!r.ok) throw new Error(`cloudflared download failed: HTTP ${r.status}`);
76
83
 
77
84
  if (asset.tgz) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openzoo",
3
- "version": "0.48.22",
3
+ "version": "0.48.23",
4
4
  "description": "Local x402-paying proxy + MCP server for openzoo.fun — point any OpenAI-compatible harness (Cursor, Claude Code, aider, SDKs) at localhost and it pays per call from a local burner wallet. Solana and Base rails live; Robinhood experimental.",
5
5
  "license": "MIT",
6
6
  "type": "module",