@enderfga/claw-orchestrator 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +218 -0
  3. package/assets/banner.jpg +0 -0
  4. package/configs/council-reviewer-prompt.md +82 -0
  5. package/configs/council-system-prompt.md +141 -0
  6. package/dist/bin/cli.d.ts +13 -0
  7. package/dist/bin/cli.js +460 -0
  8. package/dist/bin/cli.js.map +1 -0
  9. package/dist/src/base-oneshot-session.d.ts +87 -0
  10. package/dist/src/base-oneshot-session.js +228 -0
  11. package/dist/src/base-oneshot-session.js.map +1 -0
  12. package/dist/src/circuit-breaker.d.ts +21 -0
  13. package/dist/src/circuit-breaker.js +49 -0
  14. package/dist/src/circuit-breaker.js.map +1 -0
  15. package/dist/src/consensus.d.ts +20 -0
  16. package/dist/src/consensus.js +52 -0
  17. package/dist/src/consensus.js.map +1 -0
  18. package/dist/src/constants.d.ts +129 -0
  19. package/dist/src/constants.js +138 -0
  20. package/dist/src/constants.js.map +1 -0
  21. package/dist/src/council.d.ts +67 -0
  22. package/dist/src/council.js +914 -0
  23. package/dist/src/council.js.map +1 -0
  24. package/dist/src/embedded-server.d.ts +25 -0
  25. package/dist/src/embedded-server.js +360 -0
  26. package/dist/src/embedded-server.js.map +1 -0
  27. package/dist/src/inbox-manager.d.ts +38 -0
  28. package/dist/src/inbox-manager.js +111 -0
  29. package/dist/src/inbox-manager.js.map +1 -0
  30. package/dist/src/index.d.ts +63 -0
  31. package/dist/src/index.js +973 -0
  32. package/dist/src/index.js.map +1 -0
  33. package/dist/src/logger.d.ts +16 -0
  34. package/dist/src/logger.js +44 -0
  35. package/dist/src/logger.js.map +1 -0
  36. package/dist/src/models.d.ts +69 -0
  37. package/dist/src/models.js +299 -0
  38. package/dist/src/models.js.map +1 -0
  39. package/dist/src/openai-compat.d.ts +224 -0
  40. package/dist/src/openai-compat.js +756 -0
  41. package/dist/src/openai-compat.js.map +1 -0
  42. package/dist/src/persistent-codex-app-session.d.ts +108 -0
  43. package/dist/src/persistent-codex-app-session.js +465 -0
  44. package/dist/src/persistent-codex-app-session.js.map +1 -0
  45. package/dist/src/persistent-codex-session.d.ts +37 -0
  46. package/dist/src/persistent-codex-session.js +208 -0
  47. package/dist/src/persistent-codex-session.js.map +1 -0
  48. package/dist/src/persistent-cursor-session.d.ts +21 -0
  49. package/dist/src/persistent-cursor-session.js +241 -0
  50. package/dist/src/persistent-cursor-session.js.map +1 -0
  51. package/dist/src/persistent-custom-session.d.ts +78 -0
  52. package/dist/src/persistent-custom-session.js +938 -0
  53. package/dist/src/persistent-custom-session.js.map +1 -0
  54. package/dist/src/persistent-gemini-session.d.ts +21 -0
  55. package/dist/src/persistent-gemini-session.js +216 -0
  56. package/dist/src/persistent-gemini-session.js.map +1 -0
  57. package/dist/src/persistent-session.d.ts +80 -0
  58. package/dist/src/persistent-session.js +745 -0
  59. package/dist/src/persistent-session.js.map +1 -0
  60. package/dist/src/proxy/anthropic-adapter.d.ts +136 -0
  61. package/dist/src/proxy/anthropic-adapter.js +392 -0
  62. package/dist/src/proxy/anthropic-adapter.js.map +1 -0
  63. package/dist/src/proxy/handler.d.ts +39 -0
  64. package/dist/src/proxy/handler.js +365 -0
  65. package/dist/src/proxy/handler.js.map +1 -0
  66. package/dist/src/proxy/schema-cleaner.d.ts +11 -0
  67. package/dist/src/proxy/schema-cleaner.js +34 -0
  68. package/dist/src/proxy/schema-cleaner.js.map +1 -0
  69. package/dist/src/proxy/thought-cache.d.ts +19 -0
  70. package/dist/src/proxy/thought-cache.js +53 -0
  71. package/dist/src/proxy/thought-cache.js.map +1 -0
  72. package/dist/src/session-manager.d.ts +317 -0
  73. package/dist/src/session-manager.js +1528 -0
  74. package/dist/src/session-manager.js.map +1 -0
  75. package/dist/src/types.d.ts +513 -0
  76. package/dist/src/types.js +8 -0
  77. package/dist/src/types.js.map +1 -0
  78. package/dist/src/validation.d.ts +31 -0
  79. package/dist/src/validation.js +104 -0
  80. package/dist/src/validation.js.map +1 -0
  81. package/openclaw.plugin.json +122 -0
  82. package/package.json +84 -0
  83. package/skills/SKILL.md +184 -0
  84. package/skills/references/claude-cli-tracking.md +25 -0
  85. package/skills/references/cli.md +187 -0
  86. package/skills/references/council.md +210 -0
  87. package/skills/references/getting-started.md +133 -0
  88. package/skills/references/inbox.md +81 -0
  89. package/skills/references/multi-engine.md +382 -0
  90. package/skills/references/openai-compat.md +203 -0
  91. package/skills/references/sessions.md +191 -0
  92. package/skills/references/tools.md +418 -0
  93. package/skills/references/ultra.md +126 -0
@@ -0,0 +1,756 @@
1
+ /**
2
+ * OpenAI-compatible /v1/chat/completions endpoint.
3
+ *
4
+ * Bridges OpenAI API format to persistent Claude Code sessions, enabling
5
+ * webchat frontends (ChatGPT-Next-Web, Open WebUI, etc.) to use the plugin
6
+ * as a drop-in backend. Stateful sessions maximize Anthropic prompt caching.
7
+ */
8
+ import * as http from 'node:http';
9
+ import * as fs from 'node:fs';
10
+ import * as path from 'node:path';
11
+ import * as os from 'node:os';
12
+ import { randomUUID, createHash } from 'node:crypto';
13
+ import { resolveEngineAndModel } from './models.js';
14
+ import { OPENAI_COMPAT_DEFAULT_MODEL, OPENAI_COMPAT_AUTO_COMPACT_THRESHOLD, OPENAI_COMPAT_SESSION_PREFIX, } from './constants.js';
15
+ // ─── Session Key Resolution ──────────────────────────────────────────────────
16
+ /**
17
+ * Derive a session key from the request.
18
+ * Priority: X-Session-Id header > user field > sha1(model + systemPrompt) > "default"
19
+ *
20
+ * The system-prompt-hash fallback prevents the bug where every caller without
21
+ * X-Session-Id or `user` collapses onto a single shared "openai-default"
22
+ * plugin session. In multi-caller setups (OpenClaw routing the main agent,
23
+ * cron jobs, and subagents through the same gateway) that previously meant
24
+ * every request serialized against every other and frequently picked up the
25
+ * wrong session's appendSystemPrompt — also a privacy leak across callers.
26
+ *
27
+ * The model is mixed into the hash so that two callers with the same system
28
+ * prompt but different requested models don't collide and silently get
29
+ * responses from the wrong model. Originally diagnosed in PR #40 by
30
+ * @megayounus786.
31
+ */
32
+ /**
33
+ * When set (to '1', 'true', 'yes'), the proxy preserves the pre-fix behavior:
34
+ * - tools injected into every user message
35
+ * - session key NOT fingerprinted by tools (same session across tool changes)
36
+ * Default (unset) is the new behavior: tools embedded in session system prompt
37
+ * at create time + session key fingerprinted by tools. The new behavior
38
+ * eliminates periodic latency spikes but does not support mutating the tool
39
+ * list within a single session (a new session is created when tools change).
40
+ */
41
+ export function isToolsPerMessageModeEnabled() {
42
+ const v = process.env.OPENAI_COMPAT_TOOLS_PER_MESSAGE;
43
+ if (!v)
44
+ return false;
45
+ const t = v.trim().toLowerCase();
46
+ return t === '1' || t === 'true' || t === 'yes';
47
+ }
48
+ /**
49
+ * Generate the "no built-in tools" system prompt preamble.
50
+ * The `toolLocation` parameter controls how the model is told where to find
51
+ * tool definitions — 'system' means "in the <available_tools> block below"
52
+ * (tools baked into system prompt), 'user' means "in <available_tools> tags
53
+ * in the user message" (legacy per-turn injection).
54
+ */
55
+ export function noToolsSystemPrompt(toolLocation) {
56
+ const locationHint = toolLocation === 'system'
57
+ ? 'in the <available_tools> block below'
58
+ : 'in <available_tools> tags in the user message';
59
+ return ('You are a helpful AI assistant acting as a pure LLM behind an API proxy.\n' +
60
+ 'You do NOT have access to any tools such as Bash, Read, Write, Edit, Glob, Grep, or any other built-in tools.\n' +
61
+ 'Do NOT attempt to call any tools or execute any commands.\n' +
62
+ `When you need to perform an action, use ONLY the tools defined ${locationHint}, ` +
63
+ 'and respond with <tool_calls> tags as instructed there.\n' +
64
+ 'If no <available_tools> are provided, respond with text only.');
65
+ }
66
+ /**
67
+ * Build the full session system prompt for a Claude Code session with tools.
68
+ * Exported for testability — called from `handleChatCompletion`.
69
+ *
70
+ * - Default mode: tools are embedded in the system prompt (cacheable by Anthropic).
71
+ * - Legacy mode (OPENAI_COMPAT_TOOLS_PER_MESSAGE=1): tools are NOT embedded;
72
+ * they'll be injected per-turn in the user message instead.
73
+ */
74
+ export function buildSessionSystemPrompt(tools, callerSystemPrompt) {
75
+ if (isToolsPerMessageModeEnabled()) {
76
+ const preamble = noToolsSystemPrompt('user');
77
+ return callerSystemPrompt ? `${preamble}\n\n${callerSystemPrompt}` : preamble;
78
+ }
79
+ const preamble = noToolsSystemPrompt('system');
80
+ const toolBlock = buildToolPromptBlock(tools);
81
+ const systemWithTools = `${preamble}\n\n${toolBlock}`;
82
+ return callerSystemPrompt ? `${systemWithTools}\n\n${callerSystemPrompt}` : systemWithTools;
83
+ }
84
+ export function resolveSessionKey(body, headers) {
85
+ const headerKey = headers['x-session-id'];
86
+ if (typeof headerKey === 'string' && headerKey.trim())
87
+ return headerKey.trim();
88
+ if (body.user && body.user.trim())
89
+ return body.user.trim();
90
+ const sys = (body.messages || [])
91
+ .filter((m) => m && m.role === 'system')
92
+ .map((m) => (typeof m.content === 'string' ? m.content : JSON.stringify(m.content)))
93
+ .join('\n');
94
+ const modelTag = (body.model || '').toString();
95
+ // Include a fingerprint of the tool list so that two requests with the same
96
+ // system prompt but different tool definitions land in different sessions.
97
+ // The tool schemas are baked into the session system prompt on create; if
98
+ // tools change we need a new session rather than re-using a stale one.
99
+ // Hash only tool names + a short description prefix to keep the fingerprint
100
+ // small and stable against schema formatting differences.
101
+ //
102
+ // Opt-out: OPENAI_COMPAT_TOOLS_PER_MESSAGE=1 restores the pre-fix behavior
103
+ // of keying sessions only by system prompt + model. Enable this if you have
104
+ // callers that mutate their tool list within one conversation and rely on
105
+ // continuing history across tool changes.
106
+ const toolsFingerprint = isToolsPerMessageModeEnabled()
107
+ ? ''
108
+ : (body.tools || [])
109
+ .map((t) => {
110
+ const fn = t?.function;
111
+ if (!fn?.name)
112
+ return '';
113
+ const descPrefix = (typeof fn.description === 'string' ? fn.description : '').slice(0, 64);
114
+ return `${fn.name}:${descPrefix}`;
115
+ })
116
+ .filter(Boolean)
117
+ .join('|');
118
+ if (sys || modelTag || toolsFingerprint) {
119
+ return ('sys-' +
120
+ createHash('sha1')
121
+ .update(modelTag + '\n' + sys + '\n' + toolsFingerprint)
122
+ .digest('hex')
123
+ .slice(0, 12));
124
+ }
125
+ return 'default';
126
+ }
127
+ /** Build the full session name from a key */
128
+ export function sessionNameFromKey(key) {
129
+ return `${OPENAI_COMPAT_SESSION_PREFIX}${key}`;
130
+ }
131
+ // ─── Function Calling Support ────────────────────────────────────────────────
132
+ /**
133
+ * Convert OpenAI tool definitions into a structured prompt block.
134
+ * Injected into the user message so the CLI model sees tool definitions
135
+ * and responds with <tool_calls> tags when it wants to invoke a function.
136
+ */
137
+ export function buildToolPromptBlock(tools) {
138
+ if (!tools?.length)
139
+ return '';
140
+ const toolDefs = tools
141
+ .map((t) => {
142
+ const fn = t.function;
143
+ const params = JSON.stringify(fn.parameters, null, 2);
144
+ return `### ${fn.name}\n${fn.description}\n\nParameters:\n\`\`\`json\n${params}\n\`\`\``;
145
+ })
146
+ .join('\n\n');
147
+ return ('<available_tools>\n' +
148
+ 'You have access to the following tools. When you need to use a tool, respond with a JSON array wrapped in <tool_calls> tags.\n\n' +
149
+ 'FORMAT:\n' +
150
+ '<tool_calls>\n' +
151
+ '[{"name": "tool_name", "arguments": {"param1": "value1"}}]\n' +
152
+ '</tool_calls>\n\n' +
153
+ 'If you do NOT need any tools, respond normally with text only (no <tool_calls> tags).\n\n' +
154
+ '## Available Tools\n\n' +
155
+ toolDefs +
156
+ '\n</available_tools>');
157
+ }
158
+ /**
159
+ * Parse tool_calls from CLI text output.
160
+ *
161
+ * Looks for <tool_calls>[...]</tool_calls> tags in the response text.
162
+ * Returns both the extracted text content (before/after tags) and any tool calls found.
163
+ */
164
+ export function parseToolCallsFromText(text) {
165
+ // Match ALL <tool_calls> blocks (model may output multiple)
166
+ const tagRegex = /<tool_calls>\s*([\s\S]*?)\s*<\/tool_calls>/g;
167
+ const allCalls = [];
168
+ let lastIndex = 0;
169
+ const textParts = [];
170
+ let m;
171
+ while ((m = tagRegex.exec(text)) !== null) {
172
+ // Collect text before this block
173
+ const before = text.slice(lastIndex, m.index).trim();
174
+ if (before)
175
+ textParts.push(before);
176
+ lastIndex = m.index + m[0].length;
177
+ try {
178
+ const parsed = JSON.parse(m[1].trim());
179
+ const arr = Array.isArray(parsed) ? parsed : [parsed];
180
+ for (const raw of arr) {
181
+ const call = raw;
182
+ if (!call || typeof call !== 'object' || typeof call.name !== 'string')
183
+ continue;
184
+ let args;
185
+ if (typeof call.arguments === 'string') {
186
+ try {
187
+ JSON.parse(call.arguments);
188
+ args = call.arguments;
189
+ }
190
+ catch {
191
+ args = JSON.stringify({ input: call.arguments });
192
+ }
193
+ }
194
+ else {
195
+ args = JSON.stringify(call.arguments ?? {});
196
+ }
197
+ allCalls.push({
198
+ id: `call_${randomUUID().replace(/-/g, '').slice(0, 24)}`,
199
+ type: 'function',
200
+ function: { name: call.name, arguments: args },
201
+ });
202
+ }
203
+ }
204
+ catch {
205
+ // One block failed — keep its text as content
206
+ textParts.push(m[0]);
207
+ }
208
+ }
209
+ // Collect text after last block
210
+ const after = text.slice(lastIndex).trim();
211
+ if (after)
212
+ textParts.push(after);
213
+ // Strip <tool_result> and <tool_results> tags that the model may echo back
214
+ // from the serialized tool results we injected earlier.
215
+ const stripToolResultTags = (s) => s
216
+ .replace(/<tool_results?>[\s\S]*?<\/tool_results?>/g, '')
217
+ .replace(/<tool_results?[^>]*>/g, '')
218
+ .trim();
219
+ if (allCalls.length > 0) {
220
+ const raw = textParts.join('\n').trim();
221
+ const cleaned = raw ? stripToolResultTags(raw) : null;
222
+ return { textContent: cleaned || null, toolCalls: allCalls };
223
+ }
224
+ const cleaned = text ? stripToolResultTags(text) : null;
225
+ return { textContent: cleaned || null, toolCalls: [] };
226
+ }
227
+ /**
228
+ * Serialize tool result messages into a text block for the CLI model.
229
+ * Converts OpenAI `tool` role messages into <tool_result> tags.
230
+ */
231
+ export function serializeToolResults(messages) {
232
+ const toolMessages = messages.filter((m) => m.role === 'tool');
233
+ if (!toolMessages.length)
234
+ return '';
235
+ const results = toolMessages
236
+ .map((m) => {
237
+ const content = typeof m.content === 'string' ? m.content : JSON.stringify(m.content);
238
+ return `<tool_result tool_call_id="${m.tool_call_id || 'unknown'}">\n${content}\n</tool_result>`;
239
+ })
240
+ .join('\n\n');
241
+ return `<tool_results>\n${results}\n</tool_results>\n\nAbove are the results of the tool calls you requested. Continue your response based on these results.`;
242
+ }
243
+ /**
244
+ * Extract the relevant parts from an OpenAI messages array.
245
+ *
246
+ * Sessions are stateful — we only need the last user message. The tricky
247
+ * question is whether to start a fresh session or append to the existing one.
248
+ *
249
+ * Default mode (no env var): only honor an explicit `X-Session-Reset: 1`
250
+ * header. This is correct for clients that maintain their own conversation
251
+ * transcript and forward only the latest user turn (OpenClaw main agent
252
+ * loop, cron jobs, subagents). The previous heuristic
253
+ * (`nonSystemMessages.length <= 1`) fired on every such request, killing the
254
+ * persistent CLI every turn and preventing Anthropic prompt caching from
255
+ * ever warming. Originally diagnosed in PR #40 by @megayounus786.
256
+ *
257
+ * Legacy mode (`OPENAI_COMPAT_NEW_CONVO_HEURISTIC=1`): restore the old
258
+ * `system + single user ⇒ new conversation` rule, for clients that re-send
259
+ * the full transcript on every turn (ChatGPT-Next-Web, Open WebUI, data
260
+ * labeling tools, etc). They use the transcript shape itself as their only
261
+ * "start a new conversation" signal.
262
+ *
263
+ * The env var is read on every call so ops can flip it via launchctl setenv
264
+ * without restarting the server.
265
+ */
266
+ export function extractUserMessage(messages, headers) {
267
+ if (!messages || messages.length === 0) {
268
+ throw new Error('messages array is empty');
269
+ }
270
+ // Normalize content from any message: OpenAI API allows content as a string
271
+ // OR an array of content parts (e.g. multimodal messages with text + images).
272
+ // We need a string for the CLI, so arrays are joined.
273
+ const textOf = (m) => {
274
+ if (typeof m.content === 'string')
275
+ return m.content;
276
+ if (Array.isArray(m.content)) {
277
+ return m.content
278
+ .map((p) => p.text || '')
279
+ .filter(Boolean)
280
+ .join('');
281
+ }
282
+ return m.content != null ? String(m.content) : '';
283
+ };
284
+ // Extract system prompt if present
285
+ const systemMessages = messages.filter((m) => m.role === 'system');
286
+ const systemPrompt = systemMessages.length > 0 ? systemMessages.map(textOf).join('\n') : undefined;
287
+ // Handle tool result messages — only when the LAST non-system message is
288
+ // a tool role (meaning we're in an active tool-use cycle). If the last
289
+ // message is a user role, it's a follow-up in an existing conversation
290
+ // and the old tool results are already in the CLI's history.
291
+ const lastNonSystem = [...messages].reverse().find((m) => m.role !== 'system');
292
+ if (lastNonSystem?.role === 'tool') {
293
+ const toolResultBlock = serializeToolResults(messages);
294
+ const userMessages = messages.filter((m) => m.role === 'user');
295
+ const lastUserText = userMessages.length > 0 ? textOf(userMessages[userMessages.length - 1]) : '';
296
+ const userMessage = lastUserText ? `${toolResultBlock}\n\n${lastUserText}` : toolResultBlock;
297
+ return { systemPrompt, userMessage, isNewConversation: false };
298
+ }
299
+ // Find last user message
300
+ const userMessages = messages.filter((m) => m.role === 'user');
301
+ if (userMessages.length === 0) {
302
+ throw new Error('No user message found in messages array');
303
+ }
304
+ const userMessage = textOf(userMessages[userMessages.length - 1]);
305
+ // 1. Explicit reset header — honored in both modes. Normalize trim+lowercase
306
+ // so callers using `TRUE`, ` 1 `, etc. don't silently fail.
307
+ const rawReset = headers?.['x-session-reset'];
308
+ const resetHeader = typeof rawReset === 'string' ? rawReset.trim().toLowerCase() : '';
309
+ if (resetHeader === 'true' || resetHeader === '1') {
310
+ return { systemPrompt, userMessage, isNewConversation: true };
311
+ }
312
+ // 2. Legacy heuristic — only when explicitly opted in via env var.
313
+ if (process.env.OPENAI_COMPAT_NEW_CONVO_HEURISTIC === '1') {
314
+ const nonSystemMessages = messages.filter((m) => m.role !== 'system');
315
+ return { systemPrompt, userMessage, isNewConversation: nonSystemMessages.length <= 1 };
316
+ }
317
+ return { systemPrompt, userMessage, isNewConversation: false };
318
+ }
319
+ // ─── Response Formatting ─────────────────────────────────────────────────────
320
+ export function formatCompletionResponse(id, model, text, tokensIn, tokensOut, toolCalls) {
321
+ const hasToolCalls = toolCalls && toolCalls.length > 0;
322
+ return {
323
+ id,
324
+ object: 'chat.completion',
325
+ created: Math.floor(Date.now() / 1000),
326
+ model,
327
+ choices: [
328
+ {
329
+ index: 0,
330
+ message: {
331
+ role: 'assistant',
332
+ content: text || null,
333
+ ...(hasToolCalls ? { tool_calls: toolCalls } : {}),
334
+ },
335
+ finish_reason: hasToolCalls ? 'tool_calls' : 'stop',
336
+ },
337
+ ],
338
+ usage: {
339
+ prompt_tokens: tokensIn,
340
+ completion_tokens: tokensOut,
341
+ total_tokens: tokensIn + tokensOut,
342
+ },
343
+ };
344
+ }
345
+ export function formatCompletionChunk(id, model, delta, finishReason) {
346
+ return {
347
+ id,
348
+ object: 'chat.completion.chunk',
349
+ created: Math.floor(Date.now() / 1000),
350
+ model,
351
+ choices: [{ index: 0, delta, finish_reason: finishReason }],
352
+ };
353
+ }
354
+ export async function handleChatCompletion(manager, body, headers, res) {
355
+ // Validate before casting
356
+ if (!body.messages || !Array.isArray(body.messages) || body.messages.length === 0) {
357
+ res.writeHead(400, { 'Content-Type': 'application/json' });
358
+ res.end(JSON.stringify({
359
+ error: { message: 'messages is required and must be a non-empty array', type: 'invalid_request_error' },
360
+ }));
361
+ return;
362
+ }
363
+ // Safe cast: messages validated above, other fields are optional
364
+ const request = {
365
+ messages: body.messages,
366
+ model: body.model,
367
+ stream: body.stream,
368
+ temperature: body.temperature,
369
+ max_tokens: body.max_tokens,
370
+ user: body.user,
371
+ tools: body.tools,
372
+ };
373
+ // Validate max_tokens if provided
374
+ if (request.max_tokens !== undefined && (typeof request.max_tokens !== 'number' || request.max_tokens <= 0)) {
375
+ res.writeHead(400, { 'Content-Type': 'application/json' });
376
+ res.end(JSON.stringify({
377
+ error: { message: 'max_tokens must be a positive number', type: 'invalid_request_error' },
378
+ }));
379
+ return;
380
+ }
381
+ const modelStr = request.model || OPENAI_COMPAT_DEFAULT_MODEL;
382
+ const { engine, model: resolvedModel } = resolveEngineAndModel(modelStr);
383
+ const sessionKey = resolveSessionKey(request, headers);
384
+ const sessionName = sessionNameFromKey(sessionKey);
385
+ const isStreaming = request.stream === true;
386
+ let extracted;
387
+ try {
388
+ extracted = extractUserMessage(request.messages, headers);
389
+ }
390
+ catch (err) {
391
+ res.writeHead(400, { 'Content-Type': 'application/json' });
392
+ res.end(JSON.stringify({ error: { message: err.message, type: 'invalid_request_error' } }));
393
+ return;
394
+ }
395
+ // Check if session exists
396
+ const existingSessions = manager.listSessions().map((s) => s.name);
397
+ const sessionExists = existingSessions.includes(sessionName);
398
+ // If new conversation detected and session exists, stop old one first
399
+ if (extracted.isNewConversation && sessionExists) {
400
+ try {
401
+ await manager.stopSession(sessionName);
402
+ }
403
+ catch {
404
+ /* session may have already been cleaned up */
405
+ }
406
+ }
407
+ // Create session if needed
408
+ const needsCreate = !sessionExists || extracted.isNewConversation;
409
+ if (needsCreate) {
410
+ // OpenAI-compat sessions are API proxies, not coding sessions.
411
+ // Use a neutral empty temp dir so the CLI doesn't load CLAUDE.md,
412
+ // git state, or project context from wherever `serve` was started.
413
+ const sessionCwd = path.join(os.tmpdir(), `openclaw-compat-${sessionName}`);
414
+ if (!fs.existsSync(sessionCwd))
415
+ fs.mkdirSync(sessionCwd, { recursive: true });
416
+ const sessionConfig = {
417
+ name: sessionName,
418
+ cwd: sessionCwd,
419
+ engine,
420
+ model: resolvedModel,
421
+ permissionMode: 'bypassPermissions',
422
+ // skipPersistence: tells SessionManager not to write this session to
423
+ // the disk registry, preventing auto-resume of stale sessions.
424
+ // Note: noSessionPersistence (--no-session-persistence) is NOT set
425
+ // because some CLI forks don't support this flag.
426
+ skipPersistence: true,
427
+ };
428
+ // When the caller provides tool definitions, disable CLI built-in tools
429
+ // (Bash, Read, Edit, etc.) so the model uses our text-defined tools
430
+ // instead. Only works on Claude Code; forks that don't support --tools ""
431
+ // will fall back to prompt-only instructions.
432
+ if (request.tools?.length && engine === 'claude') {
433
+ sessionConfig.tools = '';
434
+ }
435
+ // Claude Code CLI supports --system-prompt (replace) and --append-system-prompt (append).
436
+ // When the caller provides tools, use --system-prompt to REPLACE the CLI's entire
437
+ // system prompt via buildSessionSystemPrompt(). See that function's doc for details
438
+ // on default vs legacy (OPENAI_COMPAT_TOOLS_PER_MESSAGE=1) behavior.
439
+ if (engine === 'claude') {
440
+ if (request.tools?.length) {
441
+ sessionConfig.systemPrompt = buildSessionSystemPrompt(request.tools, extracted.systemPrompt);
442
+ }
443
+ else if (extracted.systemPrompt) {
444
+ sessionConfig.appendSystemPrompt = extracted.systemPrompt;
445
+ }
446
+ }
447
+ try {
448
+ await manager.startSession(sessionConfig);
449
+ }
450
+ catch (err) {
451
+ res.writeHead(503, { 'Content-Type': 'application/json' });
452
+ res.end(JSON.stringify({
453
+ error: { message: `Failed to start session: ${err.message}`, type: 'server_error' },
454
+ }));
455
+ return;
456
+ }
457
+ }
458
+ // Auto-compact if context is getting full
459
+ if (sessionExists && !needsCreate) {
460
+ try {
461
+ const status = manager.getStatus(sessionName);
462
+ if (status.stats.contextPercent > OPENAI_COMPAT_AUTO_COMPACT_THRESHOLD) {
463
+ await manager.compactSession(sessionName);
464
+ }
465
+ }
466
+ catch {
467
+ /* best effort — session may not support compact */
468
+ }
469
+ }
470
+ // For non-claude engines (Cursor, Codex, Gemini), their CLIs don't support
471
+ // --append-system-prompt. Prepend the upstream system prompt to the user
472
+ // message on EVERY turn so the model sees the caller's identity, tool
473
+ // definitions, and workspace context. This is done here (not at session
474
+ // creation) because these engines spawn a fresh CLI process per turn —
475
+ // there's no persistent session to carry the system prompt forward.
476
+ let userMessage = extracted.userMessage;
477
+ if (extracted.systemPrompt && engine !== 'claude') {
478
+ userMessage = `<system>\n${extracted.systemPrompt}\n</system>\n\n${userMessage}`;
479
+ }
480
+ // Inject tool definitions into the user message.
481
+ //
482
+ // Default path for Claude Code: tools are already embedded in the session
483
+ // system prompt (see session create block above) — do NOT re-inject them
484
+ // per turn. Repeatedly prepending a large <available_tools> block to every
485
+ // user message bloats each turn's input, defeats Anthropic prompt caching,
486
+ // and was the cause of periodic 30-50s latency spikes.
487
+ //
488
+ // Opt-out path for Claude Code (OPENAI_COMPAT_TOOLS_PER_MESSAGE=1): fall
489
+ // back to the legacy behavior of injecting the tool block into each user
490
+ // message. Enables dynamic tool list updates within a single session.
491
+ //
492
+ // Non-claude engines: the CLI is spawned fresh per turn with no persistent
493
+ // system prompt, so tools must always be injected per message.
494
+ const hasTools = !!request.tools?.length;
495
+ const injectToolsPerTurn = hasTools && (engine !== 'claude' || isToolsPerMessageModeEnabled());
496
+ if (injectToolsPerTurn) {
497
+ const toolBlock = buildToolPromptBlock(request.tools);
498
+ userMessage = `${toolBlock}\n\n${userMessage}`;
499
+ }
500
+ const completionId = `chatcmpl-${randomUUID().replace(/-/g, '').slice(0, 29)}`;
501
+ if (isStreaming) {
502
+ await handleStreaming(manager, sessionName, resolvedModel, userMessage, completionId, res, hasTools);
503
+ }
504
+ else {
505
+ await handleNonStreaming(manager, sessionName, resolvedModel, userMessage, completionId, res, hasTools);
506
+ }
507
+ // Clean up ephemeral sessions immediately after response.
508
+ // When X-Session-Reset is set, each request creates a fresh session that
509
+ // should not persist — leaving it alive leaks CLI subprocesses until TTL.
510
+ if (extracted.isNewConversation) {
511
+ manager.stopSession(sessionName).catch(() => { });
512
+ }
513
+ }
514
+ // ─── Status Reporting ───────────────────────────────────────────────────────
515
+ // Push tool/thinking status to an external webhook so a webchat status bar
516
+ // can show what the CLI agent is doing. Best-effort fire-and-forget.
517
+ /**
518
+ * Optional status webhook — set `OPENAI_COMPAT_STATUS_URL` to an HTTP endpoint
519
+ * that accepts `POST { state, activity, tool }`. The bridge will fire-and-forget
520
+ * status updates when the CLI agent uses tools, so an external dashboard (e.g.
521
+ * a webchat status bar) can show real-time progress.
522
+ *
523
+ * Example: `OPENAI_COMPAT_STATUS_URL=http://127.0.0.1:18795/my-app/agent-status`
524
+ */
525
+ function reportStatus(state, activity, tool) {
526
+ const url = process.env.OPENAI_COMPAT_STATUS_URL;
527
+ if (!url)
528
+ return;
529
+ const payload = JSON.stringify({ state, activity, tool: tool || null });
530
+ const req = http.request(url, {
531
+ method: 'POST',
532
+ headers: { 'Content-Type': 'application/json', 'Content-Length': Buffer.byteLength(payload) },
533
+ timeout: 2000,
534
+ }, () => { });
535
+ req.on('error', () => { });
536
+ req.write(payload);
537
+ req.end();
538
+ }
539
+ function getToolDescription(toolName, toolInput) {
540
+ switch (toolName) {
541
+ case 'Bash':
542
+ case 'exec': {
543
+ const cmd = String(toolInput?.command || '');
544
+ return `Running: ${cmd.length > 50 ? cmd.slice(0, 50) + '...' : cmd}`;
545
+ }
546
+ case 'Read':
547
+ case 'read':
548
+ return `Reading: ${String(toolInput?.file_path || toolInput?.path || 'file')
549
+ .split('/')
550
+ .pop()}`;
551
+ case 'Write':
552
+ case 'write':
553
+ return `Writing: ${String(toolInput?.file_path || toolInput?.path || 'file')
554
+ .split('/')
555
+ .pop()}`;
556
+ case 'Edit':
557
+ case 'edit':
558
+ return `Editing: ${String(toolInput?.file_path || toolInput?.path || 'file')
559
+ .split('/')
560
+ .pop()}`;
561
+ case 'Glob':
562
+ case 'glob':
563
+ return `Searching files: ${String(toolInput?.pattern || '')}`;
564
+ case 'Grep':
565
+ case 'grep':
566
+ return `Searching content: ${String(toolInput?.pattern || '')}`;
567
+ case 'WebSearch':
568
+ return `Web search: ${String(toolInput?.query || '')}`;
569
+ case 'Agent':
570
+ return `Spawning sub-agent...`;
571
+ default:
572
+ return `Using tool: ${toolName}`;
573
+ }
574
+ }
575
+ // ─── Non-Streaming ───────────────────────────────────────────────────────────
576
+ async function handleNonStreaming(manager, sessionName, model, userMessage, completionId, res, hasTools) {
577
+ try {
578
+ reportStatus('thinking', 'Processing request...');
579
+ const result = await manager.sendMessage(sessionName, userMessage, {
580
+ onEvent: (event) => {
581
+ if (event.type === 'tool_use' && event.tool?.name) {
582
+ const desc = getToolDescription(event.tool.name, event.tool.input);
583
+ reportStatus('working', desc, event.tool.name);
584
+ }
585
+ },
586
+ });
587
+ reportStatus('idle', 'Ready');
588
+ let tokensIn = 0;
589
+ let tokensOut = 0;
590
+ try {
591
+ const status = manager.getStatus(sessionName);
592
+ tokensIn = status.stats.tokensIn;
593
+ tokensOut = status.stats.tokensOut;
594
+ }
595
+ catch {
596
+ /* stats unavailable */
597
+ }
598
+ // Parse tool_calls from response text when caller provided tools
599
+ if (hasTools) {
600
+ const parsed = parseToolCallsFromText(result.output);
601
+ const response = formatCompletionResponse(completionId, model, parsed.textContent ?? '', tokensIn, tokensOut, parsed.toolCalls.length > 0 ? parsed.toolCalls : undefined);
602
+ res.writeHead(200, { 'Content-Type': 'application/json' });
603
+ res.end(JSON.stringify(response));
604
+ }
605
+ else {
606
+ const response = formatCompletionResponse(completionId, model, result.output, tokensIn, tokensOut);
607
+ res.writeHead(200, { 'Content-Type': 'application/json' });
608
+ res.end(JSON.stringify(response));
609
+ }
610
+ }
611
+ catch (err) {
612
+ reportStatus('idle', 'Request failed');
613
+ res.writeHead(500, { 'Content-Type': 'application/json' });
614
+ res.end(JSON.stringify({ error: { message: err.message, type: 'server_error' } }));
615
+ }
616
+ }
617
+ // ─── Streaming ───────────────────────────────────────────────────────────────
618
+ async function handleStreaming(manager, sessionName, model, userMessage, completionId, res, hasTools) {
619
+ res.writeHead(200, {
620
+ 'Content-Type': 'text/event-stream',
621
+ 'Cache-Control': 'no-cache',
622
+ Connection: 'keep-alive',
623
+ 'X-Accel-Buffering': 'no',
624
+ });
625
+ let clientDisconnected = false;
626
+ res.on('close', () => {
627
+ clientDisconnected = true;
628
+ });
629
+ const writeSSE = (data) => {
630
+ if (!clientDisconnected) {
631
+ try {
632
+ res.write(`data: ${data}\n\n`);
633
+ }
634
+ catch {
635
+ clientDisconnected = true;
636
+ }
637
+ }
638
+ };
639
+ // Initial chunk with role
640
+ writeSSE(JSON.stringify(formatCompletionChunk(completionId, model, { role: 'assistant' }, null)));
641
+ // SSE keepalive heartbeat
642
+ const heartbeatTimer = setInterval(() => {
643
+ if (!clientDisconnected) {
644
+ try {
645
+ res.write(': keepalive\n\n');
646
+ }
647
+ catch {
648
+ clientDisconnected = true;
649
+ }
650
+ }
651
+ }, 30_000);
652
+ // When tools are present, buffer the full response to parse for tool_calls.
653
+ // Without tools, stream text chunks directly for low latency.
654
+ let bufferedText = '';
655
+ try {
656
+ reportStatus('thinking', 'Processing request...');
657
+ await manager.sendMessage(sessionName, userMessage, {
658
+ onChunk: (chunk) => {
659
+ if (hasTools) {
660
+ bufferedText += chunk;
661
+ // Send keepalive comments during buffering to prevent timeouts
662
+ }
663
+ else {
664
+ writeSSE(JSON.stringify(formatCompletionChunk(completionId, model, { content: chunk }, null)));
665
+ }
666
+ },
667
+ onEvent: (event) => {
668
+ if (event.type === 'tool_use' && event.tool?.name) {
669
+ reportStatus('working', getToolDescription(event.tool.name, event.tool.input), event.tool.name);
670
+ }
671
+ },
672
+ });
673
+ reportStatus('idle', 'Ready');
674
+ // Get token usage for final chunk
675
+ let usage;
676
+ try {
677
+ const status = manager.getStatus(sessionName);
678
+ usage = {
679
+ prompt_tokens: status.stats.tokensIn,
680
+ completion_tokens: status.stats.tokensOut,
681
+ total_tokens: status.stats.tokensIn + status.stats.tokensOut,
682
+ };
683
+ }
684
+ catch {
685
+ /* best effort */
686
+ }
687
+ if (hasTools && bufferedText) {
688
+ const parsed = parseToolCallsFromText(bufferedText);
689
+ if (parsed.toolCalls.length > 0) {
690
+ // Emit text content if any
691
+ if (parsed.textContent) {
692
+ writeSSE(JSON.stringify(formatCompletionChunk(completionId, model, { content: parsed.textContent }, null)));
693
+ }
694
+ // Emit tool_call chunks
695
+ for (let i = 0; i < parsed.toolCalls.length; i++) {
696
+ const tc = parsed.toolCalls[i];
697
+ writeSSE(JSON.stringify({
698
+ id: completionId,
699
+ object: 'chat.completion.chunk',
700
+ created: Math.floor(Date.now() / 1000),
701
+ model,
702
+ choices: [
703
+ {
704
+ index: 0,
705
+ delta: {
706
+ tool_calls: [
707
+ {
708
+ index: i,
709
+ id: tc.id,
710
+ type: 'function',
711
+ function: { name: tc.function.name, arguments: tc.function.arguments },
712
+ },
713
+ ],
714
+ },
715
+ finish_reason: null,
716
+ },
717
+ ],
718
+ }));
719
+ }
720
+ // Final chunk with tool_calls finish reason
721
+ const finalChunk = formatCompletionChunk(completionId, model, {}, 'tool_calls');
722
+ if (usage)
723
+ finalChunk.usage = usage;
724
+ writeSSE(JSON.stringify(finalChunk));
725
+ }
726
+ else {
727
+ // No tool calls — emit buffered text as content
728
+ writeSSE(JSON.stringify(formatCompletionChunk(completionId, model, { content: bufferedText }, null)));
729
+ const finalChunk = formatCompletionChunk(completionId, model, {}, 'stop');
730
+ if (usage)
731
+ finalChunk.usage = usage;
732
+ writeSSE(JSON.stringify(finalChunk));
733
+ }
734
+ }
735
+ else {
736
+ // No tools — standard finish
737
+ const finalChunk = formatCompletionChunk(completionId, model, {}, 'stop');
738
+ if (usage)
739
+ finalChunk.usage = usage;
740
+ writeSSE(JSON.stringify(finalChunk));
741
+ }
742
+ writeSSE('[DONE]');
743
+ }
744
+ catch (err) {
745
+ reportStatus('idle', 'Request failed');
746
+ writeSSE(JSON.stringify({ error: { message: err.message, type: 'server_error' } }));
747
+ writeSSE('[DONE]');
748
+ }
749
+ finally {
750
+ clearInterval(heartbeatTimer);
751
+ }
752
+ if (!clientDisconnected) {
753
+ res.end();
754
+ }
755
+ }
756
+ //# sourceMappingURL=openai-compat.js.map