@bike4mind/cli 1.0.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/bin/bike4mind-cli.mjs +0 -5
- package/dist/AgentHistoryStore-B2NEOvSW.mjs +18014 -0
- package/dist/{ApiClient-EUTyn5yu.mjs → ApiClient-D0fQ2FT6.mjs} +2 -2
- package/dist/{BubblewrapRuntime-CkL9-gnG.mjs → BubblewrapRuntime-5bLPTwEC.mjs} +2 -2
- package/dist/{ConfigStore-DHNgFdOu.mjs → ConfigStore-qV7NrCgZ.mjs} +2259 -1283
- package/dist/{ProxyManager-C1-lgzEU.mjs → ProxyManager-B0-RuR2w.mjs} +21 -4
- package/dist/{SandboxOrchestrator-C8uleDn2.mjs → SandboxOrchestrator-BcUv9fQ3.mjs} +47 -13
- package/dist/{SandboxRuntimeAdapter-ChGlxSGQ.mjs → SandboxRuntimeAdapter-BgLUVTJL.mjs} +2 -2
- package/dist/{SandboxRuntimeAdapter-CKelGICD.mjs → SandboxRuntimeAdapter-BmHELLuM.mjs} +1 -1
- package/dist/{SeatbeltRuntime-Qqt19cAN.mjs → SeatbeltRuntime-C_Y8q8Mr.mjs} +9 -1
- package/dist/buildAgent-C-C-VGff.mjs +2001 -0
- package/dist/commands/acpCommand.mjs +36 -13
- package/dist/commands/apiCommand.mjs +1 -1
- package/dist/commands/doctorCommand.mjs +1 -1
- package/dist/commands/envCommand.mjs +1 -1
- package/dist/commands/headlessCommand.mjs +92 -42
- package/dist/commands/mcpCommand.mjs +8 -13
- package/dist/commands/pluginCommand.mjs +9 -15
- package/dist/commands/updateCommand.mjs +1 -1
- package/dist/{createFile-DPv180yF-BnWFIxey.mjs → createFile-B8bur5Rb-CVzCarEA.mjs} +2 -2
- package/dist/{deleteFile-BdjUwUQF-B3XOJmg3.mjs → deleteFile-9B3gW_Nb-DG2sovIl.mjs} +2 -2
- package/dist/{globFiles-DjfDGaUK-CNR8pMRC.mjs → globFiles-CwJ8qmYo-BR5b2KvO.mjs} +3 -2
- package/dist/{grepSearch-BaYUfIYs-n0XKoGnL.mjs → grepSearch-BgoOOwGe-DtlV8Gn-.mjs} +3 -3
- package/dist/index.mjs +638 -129
- package/dist/{package-7a45-Svr.mjs → package-CGZIoxcs.mjs} +1 -1
- package/dist/{pathValidation-D8tjkQXE-1HwvsuYT.mjs → pathValidation-BRqf4HFX-CHwtwp3O.mjs} +7 -3
- package/dist/{serve-CavAHPdQ.mjs → serve-CPXqcEZr.mjs} +2 -2
- package/dist/types-CdIKgWWe.mjs +3 -0
- package/dist/{types-LyRNHOiS.mjs → types-F61_hxmG.mjs} +2 -0
- package/package.json +28 -29
- package/dist/AgentHistoryStore-CrRb8cMt.mjs +0 -38586
- package/dist/ProxyManager-B1jFWL7b.mjs +0 -3
- package/dist/SandboxOrchestrator-BFPVpmB5.mjs +0 -3
- package/dist/buildAgent-CJrkEG0M.mjs +0 -824
- package/dist/types-CqscS34o.mjs +0 -3
|
@@ -0,0 +1,2001 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { I as generateCliTools, J as setWebSocketToolExecutor, L as wrapTools, N as ReActAgent, X as buildSystemPrompt, Y as getPlanModeFilePath, a as loadProjectContext, c as createSseBackend, i as buildProjectAgentStore, it as Logger, l as ServerLlmBackend, n as BackgroundAgentManager, o as McpManager, r as SubagentOrchestrator, t as AgentHistoryStore } from "./AgentHistoryStore-B2NEOvSW.mjs";
|
|
3
|
+
import { D as MODEL_INFO_FIELD_GROUP_OF, H as escapeThinkMarkers, P as PermissionDeniedError, S as CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS, V as createThinkMarkerEscaper, W as getQuestErrorCode, Y as isUserInitiatedAbort, k as ModelBackend, n as logger, w as ChatModels } from "./ConfigStore-qV7NrCgZ.mjs";
|
|
4
|
+
import "crypto";
|
|
5
|
+
import { v4 } from "uuid";
|
|
6
|
+
import { z as z$1 } from "zod";
|
|
7
|
+
import "axios";
|
|
8
|
+
import "@aws-sdk/client-bedrock-runtime";
|
|
9
|
+
import "openai";
|
|
10
|
+
import "@anthropic-ai/sdk";
|
|
11
|
+
import "@aws-sdk/client-cloudwatch";
|
|
12
|
+
import "@google/genai";
|
|
13
|
+
import "lodash/pick.js";
|
|
14
|
+
import "openai/streaming";
|
|
15
|
+
import { Ollama } from "ollama";
|
|
16
|
+
import { Agent } from "undici";
|
|
17
|
+
import WebSocket from "ws";
|
|
18
|
+
//#region ../../b4m-core/llm-adapters/dist/index.mjs
|
|
19
|
+
/**
|
|
20
|
+
* A tool failure that must end the turn rather than be fed back to the model as a
|
|
21
|
+
* recoverable observation: a declined tool (PermissionDeniedError) or an out-of-credits
|
|
22
|
+
* error (tagged via `getQuestErrorCode`), neither of which any retry can satisfy.
|
|
23
|
+
*/
|
|
24
|
+
function isTerminalToolError(error) {
|
|
25
|
+
return error instanceof PermissionDeniedError || getQuestErrorCode(error) !== void 0;
|
|
26
|
+
}
|
|
27
|
+
/**
|
|
28
|
+
* Execute an array of async tasks either in parallel (with concurrency limiting)
|
|
29
|
+
* or sequentially, returning outcomes in the original order.
|
|
30
|
+
*
|
|
31
|
+
* Terminal errors (see isTerminalToolError) are always re-thrown immediately: sequential
|
|
32
|
+
* breaks the loop on the first one; parallel runs every task to completion (inherent to
|
|
33
|
+
* Promise.allSettled) then pre-scans outcomes before returning - accepted because these
|
|
34
|
+
* errors are rare in multi-tool batches.
|
|
35
|
+
*
|
|
36
|
+
* Fault isolation: uses Promise.allSettled so a single failure doesn't abort the batch.
|
|
37
|
+
* Order preservation: results are indexed back to the original task array.
|
|
38
|
+
*/
|
|
39
|
+
async function executeToolsBatch(tasks, options) {
|
|
40
|
+
const { parallel, maxConcurrency = 8 } = options;
|
|
41
|
+
if (parallel && tasks.length > 1) {
|
|
42
|
+
const outcomes = (await runWithConcurrency(tasks, maxConcurrency)).map((s) => s.status === "fulfilled" ? {
|
|
43
|
+
ok: true,
|
|
44
|
+
result: s.value
|
|
45
|
+
} : {
|
|
46
|
+
ok: false,
|
|
47
|
+
error: s.reason
|
|
48
|
+
});
|
|
49
|
+
for (const outcome of outcomes) if (!outcome.ok && isTerminalToolError(outcome.error)) throw outcome.error;
|
|
50
|
+
return outcomes;
|
|
51
|
+
}
|
|
52
|
+
const outcomes = [];
|
|
53
|
+
for (const task of tasks) try {
|
|
54
|
+
const result = await task();
|
|
55
|
+
outcomes.push({
|
|
56
|
+
ok: true,
|
|
57
|
+
result
|
|
58
|
+
});
|
|
59
|
+
} catch (error) {
|
|
60
|
+
if (isTerminalToolError(error)) throw error;
|
|
61
|
+
outcomes.push({
|
|
62
|
+
ok: false,
|
|
63
|
+
error
|
|
64
|
+
});
|
|
65
|
+
}
|
|
66
|
+
return outcomes;
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* Run tasks with bounded concurrency using a worker-pool pattern.
|
|
70
|
+
* Spawns up to `limit` workers that pull from a shared task queue.
|
|
71
|
+
* Results are stored by index to preserve original ordering.
|
|
72
|
+
*/
|
|
73
|
+
async function runWithConcurrency(tasks, limit) {
|
|
74
|
+
if (limit >= tasks.length) return Promise.allSettled(tasks.map((fn) => fn()));
|
|
75
|
+
const results = new Array(tasks.length);
|
|
76
|
+
let nextIndex = 0;
|
|
77
|
+
async function worker() {
|
|
78
|
+
while (nextIndex < tasks.length) {
|
|
79
|
+
const index = nextIndex++;
|
|
80
|
+
try {
|
|
81
|
+
results[index] = {
|
|
82
|
+
status: "fulfilled",
|
|
83
|
+
value: await tasks[index]()
|
|
84
|
+
};
|
|
85
|
+
} catch (reason) {
|
|
86
|
+
results[index] = {
|
|
87
|
+
status: "rejected",
|
|
88
|
+
reason
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
const workerCount = Math.min(limit, tasks.length);
|
|
94
|
+
await Promise.all(Array.from({ length: workerCount }, () => worker()));
|
|
95
|
+
return results;
|
|
96
|
+
}
|
|
97
|
+
/**
|
|
98
|
+
* Attaches a tool call's outcome onto its `toolsUsed` entry so it survives into
|
|
99
|
+
* `promptMeta.functionCalls.returnValue`/`.success` (see ChatCompletionProcess.ts's mapper
|
|
100
|
+
* and utils.ts's `replayableToolCalls`, which gates a whole replay path on at least one
|
|
101
|
+
* recorded `returnValue`). Every backend pushes a `toolsUsed` entry before executing the
|
|
102
|
+
* tool and only learns the real outcome a few lines later - this is the merge-back.
|
|
103
|
+
*/
|
|
104
|
+
/**
|
|
105
|
+
* Cap applied to a persisted `returnValue` before it reaches Mongo (chars, not bytes). Also the
|
|
106
|
+
* cap on what a later turn replays back to the model for this call (utils.ts's Priority 2
|
|
107
|
+
* reconstruction reads the same persisted, already-truncated value) - this is not a
|
|
108
|
+
* persistence-only limit, it is what a continued conversation sees of an older tool result too.
|
|
109
|
+
*/
|
|
110
|
+
const MAX_RECORDED_TOOL_RESULT_CHARS = 8e3;
|
|
111
|
+
const TOOL_RESULT_TRUNCATION_NOTICE = "\n[tool result truncated]";
|
|
112
|
+
/** Cap on the in-memory full result (chars). */
|
|
113
|
+
const MAX_FULL_TOOL_RESULT_CHARS = 2e5;
|
|
114
|
+
const FULL_TOOL_RESULT_KEY = "__b4mFullToolResult";
|
|
115
|
+
function truncateToolResult(observation) {
|
|
116
|
+
if (observation.length <= 8e3) return observation;
|
|
117
|
+
return observation.slice(0, MAX_RECORDED_TOOL_RESULT_CHARS) + TOOL_RESULT_TRUNCATION_NOTICE;
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Keeps the untruncated result on the entry for the reply parser's echo check (see
|
|
121
|
+
* buildToolEchoSources in services). It is non-enumerable so JSON.stringify and spreads skip it:
|
|
122
|
+
* `toolsUsed` also leaves the process whole (Research Mode streams each callback's completionInfo
|
|
123
|
+
* over the websocket, where API Gateway caps a post at 128 KB), and it must never be persisted.
|
|
124
|
+
*/
|
|
125
|
+
function attachFullToolResult(entry, observation) {
|
|
126
|
+
const text = String(observation);
|
|
127
|
+
const value = {
|
|
128
|
+
text: text.slice(0, MAX_FULL_TOOL_RESULT_CHARS),
|
|
129
|
+
truncated: text.length > MAX_FULL_TOOL_RESULT_CHARS
|
|
130
|
+
};
|
|
131
|
+
Object.defineProperty(entry, FULL_TOOL_RESULT_KEY, {
|
|
132
|
+
value,
|
|
133
|
+
enumerable: false,
|
|
134
|
+
configurable: true,
|
|
135
|
+
writable: true
|
|
136
|
+
});
|
|
137
|
+
}
|
|
138
|
+
/**
|
|
139
|
+
* Drops the messages whose instructions assume a tool is attached (see `IMessage.requiresTool`).
|
|
140
|
+
*
|
|
141
|
+
* Call this wherever a turn continues with tools removed. Re-sending "you MUST use the X tool" on a
|
|
142
|
+
* call that carries no tools leaves the model no valid move, and it answers by emitting the tool call
|
|
143
|
+
* as text in the reply - which is exactly what the callers that inject those prompts gate against
|
|
144
|
+
* when they decide whether to add them in the first place.
|
|
145
|
+
*
|
|
146
|
+
* Drops on the presence of the marker, not on which tool it names, because every call site today
|
|
147
|
+
* removes the whole tool set at once. A site that ever drops only SOME tools would need to compare
|
|
148
|
+
* `requiresTool` against the survivors instead.
|
|
149
|
+
*/
|
|
150
|
+
const stripToolDependentMessages = (messages) => messages.filter((message) => !message.requiresTool);
|
|
151
|
+
/**
|
|
152
|
+
* Anthropic's hard ceiling on `cache_control` markers per request. Exceeding it fails the
|
|
153
|
+
* WHOLE request with `ValidationException: A maximum of 4 blocks with cache_control may be
|
|
154
|
+
* provided`, which is non-retryable - so an over-budget request loses the turn outright,
|
|
155
|
+
* after the user has already waited for it.
|
|
156
|
+
*/
|
|
157
|
+
const MAX_CACHE_CONTROL_BLOCKS = 4;
|
|
158
|
+
/**
|
|
159
|
+
* Does this block already carry a marker? Re-marking one costs no budget.
|
|
160
|
+
*
|
|
161
|
+
* Tests the VALUE, not just key presence: a block carrying an explicit
|
|
162
|
+
* `cache_control: undefined` is not a marker as far as the provider is concerned, and counting
|
|
163
|
+
* it would spend budget on nothing and drop a breakpoint we could have kept.
|
|
164
|
+
*/
|
|
165
|
+
function hasMarker(block) {
|
|
166
|
+
return !!block && typeof block === "object" && !!block.cache_control;
|
|
167
|
+
}
|
|
168
|
+
/**
|
|
169
|
+
* Markers already on the request. Callers upstream attach their own before this runs -
|
|
170
|
+
* `bedrockBackend/anthropic.ts` marks each system block flagged `cache: true` (the mid-stack
|
|
171
|
+
* shareable-prefix breakpoint) - so this adapter's budget is whatever they left, not the full four.
|
|
172
|
+
*/
|
|
173
|
+
function censusMarkers(params) {
|
|
174
|
+
const tools = Array.isArray(params.tools) ? params.tools.filter(hasMarker).length : 0;
|
|
175
|
+
const system = Array.isArray(params.system) ? params.system.filter(hasMarker).length : 0;
|
|
176
|
+
let messages = 0;
|
|
177
|
+
if (Array.isArray(params.messages)) for (const message of params.messages) {
|
|
178
|
+
const content = message?.content;
|
|
179
|
+
if (Array.isArray(content)) messages += content.filter(hasMarker).length;
|
|
180
|
+
}
|
|
181
|
+
return {
|
|
182
|
+
tools,
|
|
183
|
+
system,
|
|
184
|
+
messages,
|
|
185
|
+
total: tools + system + messages
|
|
186
|
+
};
|
|
187
|
+
}
|
|
188
|
+
/**
|
|
189
|
+
* Anthropic-specific caching adapter
|
|
190
|
+
* Adds explicit cache_control markers to content blocks
|
|
191
|
+
*/
|
|
192
|
+
var AnthropicCachingAdapter = class {
|
|
193
|
+
applyCaching(apiParams, strategy, logger) {
|
|
194
|
+
if (!strategy.enableCaching) return apiParams;
|
|
195
|
+
const ttl = strategy.cacheTTL ?? "5m";
|
|
196
|
+
const modifiedParams = { ...apiParams };
|
|
197
|
+
const cacheControl = {
|
|
198
|
+
type: "ephemeral",
|
|
199
|
+
...ttl === "1h" ? { ttl } : {}
|
|
200
|
+
};
|
|
201
|
+
const inbound = censusMarkers(modifiedParams);
|
|
202
|
+
let budget = MAX_CACHE_CONTROL_BLOCKS - inbound.total;
|
|
203
|
+
const dropped = [];
|
|
204
|
+
/** Claim one marker slot, or record the miss. Re-marking a marked block is free. */
|
|
205
|
+
const claim = (name, alreadyMarked) => {
|
|
206
|
+
if (alreadyMarked) return true;
|
|
207
|
+
if (budget <= 0) {
|
|
208
|
+
dropped.push(name);
|
|
209
|
+
return false;
|
|
210
|
+
}
|
|
211
|
+
budget -= 1;
|
|
212
|
+
return true;
|
|
213
|
+
};
|
|
214
|
+
const systemParam = modifiedParams.system;
|
|
215
|
+
if (strategy.cacheSystemPrompt && systemParam) {
|
|
216
|
+
const systemArray = Array.isArray(systemParam) ? [...systemParam] : [{
|
|
217
|
+
type: "text",
|
|
218
|
+
text: systemParam
|
|
219
|
+
}];
|
|
220
|
+
if (systemArray.length > 0) {
|
|
221
|
+
const lastBlock = systemArray[systemArray.length - 1];
|
|
222
|
+
if (claim("system", hasMarker(lastBlock))) {
|
|
223
|
+
systemArray[systemArray.length - 1] = {
|
|
224
|
+
...lastBlock,
|
|
225
|
+
cache_control: cacheControl
|
|
226
|
+
};
|
|
227
|
+
modifiedParams.system = systemArray;
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
const messagesParam = modifiedParams.messages;
|
|
232
|
+
if (strategy.cacheConversationHistory && Array.isArray(messagesParam) && messagesParam.length > 0) {
|
|
233
|
+
const messages = [...messagesParam];
|
|
234
|
+
const anchorIndex = messages.length - 1 - (strategy.historyCacheExcludeTrailingCount ?? 0);
|
|
235
|
+
if (anchorIndex >= 0) {
|
|
236
|
+
const anchorMsg = messages[anchorIndex];
|
|
237
|
+
const msgContent = anchorMsg.content;
|
|
238
|
+
let contentArray;
|
|
239
|
+
if (typeof msgContent === "string") contentArray = [{
|
|
240
|
+
type: "text",
|
|
241
|
+
text: msgContent
|
|
242
|
+
}];
|
|
243
|
+
else if (Array.isArray(msgContent)) contentArray = [...msgContent];
|
|
244
|
+
if (contentArray && contentArray.length > 0) {
|
|
245
|
+
const lastBlock = contentArray[contentArray.length - 1];
|
|
246
|
+
if (claim("history", hasMarker(lastBlock))) {
|
|
247
|
+
contentArray[contentArray.length - 1] = {
|
|
248
|
+
...lastBlock,
|
|
249
|
+
cache_control: cacheControl
|
|
250
|
+
};
|
|
251
|
+
messages[anchorIndex] = {
|
|
252
|
+
...anchorMsg,
|
|
253
|
+
content: contentArray
|
|
254
|
+
};
|
|
255
|
+
modifiedParams.messages = messages;
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
const tools = modifiedParams.tools;
|
|
261
|
+
if (strategy.cacheTools && Array.isArray(tools) && tools.length > 0) {
|
|
262
|
+
const toolsCopy = [...tools];
|
|
263
|
+
const lastTool = toolsCopy[toolsCopy.length - 1];
|
|
264
|
+
if (claim("tools", hasMarker(lastTool))) {
|
|
265
|
+
toolsCopy[toolsCopy.length - 1] = {
|
|
266
|
+
...lastTool,
|
|
267
|
+
cache_control: cacheControl
|
|
268
|
+
};
|
|
269
|
+
modifiedParams.tools = toolsCopy;
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
const outbound = censusMarkers(modifiedParams);
|
|
273
|
+
const census = {
|
|
274
|
+
inbound,
|
|
275
|
+
outbound,
|
|
276
|
+
limit: MAX_CACHE_CONTROL_BLOCKS
|
|
277
|
+
};
|
|
278
|
+
if (outbound.total >= MAX_CACHE_CONTROL_BLOCKS) {
|
|
279
|
+
const message = "[PromptCache] cache_control census at the ceiling";
|
|
280
|
+
if (logger) logger.info(message, census);
|
|
281
|
+
else console.info(message, JSON.stringify(census));
|
|
282
|
+
} else if (logger) logger.debug("[PromptCache] cache_control census", census);
|
|
283
|
+
if (outbound.total > MAX_CACHE_CONTROL_BLOCKS) {
|
|
284
|
+
const message = `[PromptCache] request exceeds the ${MAX_CACHE_CONTROL_BLOCKS}-block cache_control limit on arrival (${outbound.total}); the provider will reject it`;
|
|
285
|
+
const detail = {
|
|
286
|
+
inbound,
|
|
287
|
+
outbound,
|
|
288
|
+
limit: MAX_CACHE_CONTROL_BLOCKS
|
|
289
|
+
};
|
|
290
|
+
if (logger) logger.error(message, detail);
|
|
291
|
+
else console.error(message, JSON.stringify(detail));
|
|
292
|
+
} else if (dropped.length > 0) {
|
|
293
|
+
const message = `[PromptCache] cache_control budget exhausted (limit ${MAX_CACHE_CONTROL_BLOCKS}); skipped breakpoints: ${dropped.join(", ")}`;
|
|
294
|
+
const detail = {
|
|
295
|
+
dropped,
|
|
296
|
+
inbound,
|
|
297
|
+
outbound,
|
|
298
|
+
limit: MAX_CACHE_CONTROL_BLOCKS
|
|
299
|
+
};
|
|
300
|
+
if (logger) logger.warn(message, detail);
|
|
301
|
+
else console.warn(message, JSON.stringify(detail));
|
|
302
|
+
}
|
|
303
|
+
return modifiedParams;
|
|
304
|
+
}
|
|
305
|
+
extractCacheStats(response, model) {
|
|
306
|
+
const usage = response.usage;
|
|
307
|
+
if (!usage) return void 0;
|
|
308
|
+
const cacheReadTokens = usage.cache_read_input_tokens || 0;
|
|
309
|
+
const cacheWriteTokens = usage.cache_creation_input_tokens || 0;
|
|
310
|
+
const uncachedTokens = usage.input_tokens || 0;
|
|
311
|
+
const totalInputTokens = cacheReadTokens + cacheWriteTokens + uncachedTokens;
|
|
312
|
+
const cacheHitRate = totalInputTokens > 0 ? cacheReadTokens / totalInputTokens * 100 : 0;
|
|
313
|
+
const costSavingsPercent = cacheHitRate * .9;
|
|
314
|
+
const estimatedLatencyReduction = cacheHitRate * .85;
|
|
315
|
+
return {
|
|
316
|
+
provider: ModelBackend.Anthropic,
|
|
317
|
+
model,
|
|
318
|
+
totalInputTokens,
|
|
319
|
+
cacheReadTokens,
|
|
320
|
+
cacheWriteTokens,
|
|
321
|
+
uncachedTokens,
|
|
322
|
+
cacheHitRate,
|
|
323
|
+
costSavingsPercent,
|
|
324
|
+
estimatedLatencyReduction,
|
|
325
|
+
providerMetadata: { hadCacheWrite: cacheWriteTokens > 0 }
|
|
326
|
+
};
|
|
327
|
+
}
|
|
328
|
+
};
|
|
329
|
+
/**
|
|
330
|
+
* OpenAI-specific caching adapter
|
|
331
|
+
* OpenAI caching is AUTOMATIC - no explicit markers needed
|
|
332
|
+
* Just ensure static content is at the beginning of prompts
|
|
333
|
+
*/
|
|
334
|
+
var OpenAICachingAdapter = class {
|
|
335
|
+
applyCaching(apiParams, strategy) {
|
|
336
|
+
return apiParams;
|
|
337
|
+
}
|
|
338
|
+
extractCacheStats(response, model) {
|
|
339
|
+
const usage = response.usage;
|
|
340
|
+
if (!usage) return void 0;
|
|
341
|
+
const totalInputTokens = usage.prompt_tokens || 0;
|
|
342
|
+
const cachedTokens = usage.prompt_tokens_details?.cached_tokens || 0;
|
|
343
|
+
const cacheHitRate = totalInputTokens > 0 ? cachedTokens / totalInputTokens * 100 : 0;
|
|
344
|
+
const costSavingsPercent = cacheHitRate * .9;
|
|
345
|
+
const estimatedLatencyReduction = cacheHitRate * .8;
|
|
346
|
+
return {
|
|
347
|
+
provider: ModelBackend.OpenAI,
|
|
348
|
+
model,
|
|
349
|
+
totalInputTokens,
|
|
350
|
+
cacheReadTokens: cachedTokens,
|
|
351
|
+
cacheWriteTokens: 0,
|
|
352
|
+
uncachedTokens: totalInputTokens - cachedTokens,
|
|
353
|
+
cacheHitRate,
|
|
354
|
+
costSavingsPercent,
|
|
355
|
+
estimatedLatencyReduction,
|
|
356
|
+
providerMetadata: { automatic: true }
|
|
357
|
+
};
|
|
358
|
+
}
|
|
359
|
+
};
|
|
360
|
+
/**
|
|
361
|
+
* Gemini-specific caching adapter
|
|
362
|
+
* Gemini 2.5+ supports IMPLICIT (automatic) caching by default
|
|
363
|
+
* Explicit caching via Vertex AI API is optional (not implemented yet)
|
|
364
|
+
*/
|
|
365
|
+
var GeminiCachingAdapter = class {
|
|
366
|
+
applyCaching(apiParams, strategy) {
|
|
367
|
+
return apiParams;
|
|
368
|
+
}
|
|
369
|
+
extractCacheStats(response, model) {
|
|
370
|
+
const usage = response.usageMetadata;
|
|
371
|
+
if (!usage) return void 0;
|
|
372
|
+
const totalInputTokens = usage.promptTokenCount || 0;
|
|
373
|
+
const cachedTokens = usage.cachedContentTokenCount || 0;
|
|
374
|
+
const cacheHitRate = totalInputTokens > 0 ? cachedTokens / totalInputTokens * 100 : 0;
|
|
375
|
+
const costSavingsPercent = cacheHitRate * (model.includes("2.5") ? .9 : .75);
|
|
376
|
+
const estimatedLatencyReduction = cacheHitRate * .75;
|
|
377
|
+
return {
|
|
378
|
+
provider: ModelBackend.Gemini,
|
|
379
|
+
model,
|
|
380
|
+
totalInputTokens,
|
|
381
|
+
cacheReadTokens: cachedTokens,
|
|
382
|
+
cacheWriteTokens: 0,
|
|
383
|
+
uncachedTokens: totalInputTokens - cachedTokens,
|
|
384
|
+
cacheHitRate,
|
|
385
|
+
costSavingsPercent,
|
|
386
|
+
estimatedLatencyReduction,
|
|
387
|
+
providerMetadata: { implicit: true }
|
|
388
|
+
};
|
|
389
|
+
}
|
|
390
|
+
};
|
|
391
|
+
/**
|
|
392
|
+
* xAI Grok-specific caching adapter
|
|
393
|
+
* Caching is automatic with optional conversation ID for cache affinity
|
|
394
|
+
*/
|
|
395
|
+
var XAICachingAdapter = class {
|
|
396
|
+
applyCaching(apiParams, strategy) {
|
|
397
|
+
return apiParams;
|
|
398
|
+
}
|
|
399
|
+
getHeaders(strategy) {
|
|
400
|
+
const headers = {};
|
|
401
|
+
if (strategy.conversationId) headers["x-grok-conv-id"] = strategy.conversationId;
|
|
402
|
+
return headers;
|
|
403
|
+
}
|
|
404
|
+
extractCacheStats(response, model) {
|
|
405
|
+
const usage = response.usage;
|
|
406
|
+
if (!usage) return void 0;
|
|
407
|
+
const totalInputTokens = usage.prompt_tokens || 0;
|
|
408
|
+
const cachedTokens = usage.cached_prompt_tokens || 0;
|
|
409
|
+
const cacheHitRate = totalInputTokens > 0 ? cachedTokens / totalInputTokens * 100 : 0;
|
|
410
|
+
const costSavingsPercent = cacheHitRate * .65;
|
|
411
|
+
const estimatedLatencyReduction = cacheHitRate * .7;
|
|
412
|
+
return {
|
|
413
|
+
provider: ModelBackend.XAI,
|
|
414
|
+
model,
|
|
415
|
+
totalInputTokens,
|
|
416
|
+
cacheReadTokens: cachedTokens,
|
|
417
|
+
cacheWriteTokens: 0,
|
|
418
|
+
uncachedTokens: totalInputTokens - cachedTokens,
|
|
419
|
+
cacheHitRate,
|
|
420
|
+
costSavingsPercent,
|
|
421
|
+
estimatedLatencyReduction,
|
|
422
|
+
providerMetadata: {
|
|
423
|
+
automatic: true,
|
|
424
|
+
conversationId: usage.x_grok_conv_id
|
|
425
|
+
}
|
|
426
|
+
};
|
|
427
|
+
}
|
|
428
|
+
};
|
|
429
|
+
/**
|
|
430
|
+
* Moonshot (Kimi) context caching. Automatic, like xAI's: there is no parameter,
|
|
431
|
+
* no header, and no explicit cache-creation call. A prompt only becomes cacheable
|
|
432
|
+
* once it exceeds 256 tokens, so short turns legitimately report a 0% hit rate.
|
|
433
|
+
* @see https://platform.kimi.ai/docs/guide/use-context-caching-feature-of-kimi-api
|
|
434
|
+
*/
|
|
435
|
+
var KimiCachingAdapter = class {
|
|
436
|
+
applyCaching(apiParams, _strategy) {
|
|
437
|
+
return apiParams;
|
|
438
|
+
}
|
|
439
|
+
extractCacheStats(response, model) {
|
|
440
|
+
const usage = response.usage;
|
|
441
|
+
if (!usage) return void 0;
|
|
442
|
+
const totalInputTokens = usage.prompt_tokens || 0;
|
|
443
|
+
const details = usage.prompt_tokens_details;
|
|
444
|
+
const cachedTokens = usage.cached_tokens || details?.cached_tokens || 0;
|
|
445
|
+
const cacheHitRate = totalInputTokens > 0 ? cachedTokens / totalInputTokens * 100 : 0;
|
|
446
|
+
const costSavingsPercent = cacheHitRate * .9;
|
|
447
|
+
const estimatedLatencyReduction = cacheHitRate * .7;
|
|
448
|
+
return {
|
|
449
|
+
provider: ModelBackend.Kimi,
|
|
450
|
+
model,
|
|
451
|
+
totalInputTokens,
|
|
452
|
+
cacheReadTokens: cachedTokens,
|
|
453
|
+
cacheWriteTokens: 0,
|
|
454
|
+
uncachedTokens: Math.max(0, totalInputTokens - cachedTokens),
|
|
455
|
+
cacheHitRate,
|
|
456
|
+
costSavingsPercent,
|
|
457
|
+
estimatedLatencyReduction,
|
|
458
|
+
providerMetadata: {
|
|
459
|
+
automatic: true,
|
|
460
|
+
minimumCacheablePromptTokens: 256
|
|
461
|
+
}
|
|
462
|
+
};
|
|
463
|
+
}
|
|
464
|
+
};
|
|
465
|
+
/**
|
|
466
|
+
* Cached prompt tokens from a raw provider usage object, across every spelling in use:
|
|
467
|
+
* OpenAI Chat Completions nests them under `prompt_tokens_details`, the OpenAI
|
|
468
|
+
* Responses API under `input_tokens_details`, Moonshot publishes a flat
|
|
469
|
+
* `cached_tokens` alongside the OpenAI-shaped nesting, and DeepSeek its own flat
|
|
470
|
+
* `prompt_cache_hit_tokens`. Reading only one spelling silently bills every cache
|
|
471
|
+
* hit on the other transports at the full input rate.
|
|
472
|
+
*
|
|
473
|
+
* DeepSeek's own spelling leads, because it is the number its invoice is computed
|
|
474
|
+
* from; the OpenAI-shaped ones it also sends are the fallback for a proxy that
|
|
475
|
+
* forwards only those.
|
|
476
|
+
*/
|
|
477
|
+
function cachedTokensFromUsage(usage) {
|
|
478
|
+
if (!usage) return 0;
|
|
479
|
+
const candidates = [
|
|
480
|
+
usage.prompt_cache_hit_tokens,
|
|
481
|
+
usage.cached_tokens,
|
|
482
|
+
usage.prompt_tokens_details?.cached_tokens,
|
|
483
|
+
usage.input_tokens_details?.cached_tokens
|
|
484
|
+
];
|
|
485
|
+
for (const value of candidates) if (typeof value === "number" && Number.isFinite(value) && value > 0) return value;
|
|
486
|
+
return 0;
|
|
487
|
+
}
|
|
488
|
+
/**
|
|
489
|
+
* DeepSeek context caching. Automatic, like Moonshot's and xAI's: no parameter,
|
|
490
|
+
* no header, no explicit cache-creation call. The adapter exists only to read
|
|
491
|
+
* the counters back out.
|
|
492
|
+
* @see https://api-docs.deepseek.com/guides/kv_cache
|
|
493
|
+
*/
|
|
494
|
+
var DeepSeekCachingAdapter = class {
|
|
495
|
+
applyCaching(apiParams, _strategy) {
|
|
496
|
+
return apiParams;
|
|
497
|
+
}
|
|
498
|
+
extractCacheStats(response, model) {
|
|
499
|
+
const usage = response.usage;
|
|
500
|
+
if (!usage) return void 0;
|
|
501
|
+
const totalInputTokens = usage.prompt_tokens || 0;
|
|
502
|
+
const cachedTokens = cachedTokensFromUsage(usage);
|
|
503
|
+
const cacheHitRate = totalInputTokens > 0 ? cachedTokens / totalInputTokens * 100 : 0;
|
|
504
|
+
const costSavingsPercent = cacheHitRate * .98;
|
|
505
|
+
const estimatedLatencyReduction = cacheHitRate * .7;
|
|
506
|
+
return {
|
|
507
|
+
provider: ModelBackend.DeepSeek,
|
|
508
|
+
model,
|
|
509
|
+
totalInputTokens,
|
|
510
|
+
cacheReadTokens: cachedTokens,
|
|
511
|
+
cacheWriteTokens: 0,
|
|
512
|
+
uncachedTokens: Math.max(0, totalInputTokens - cachedTokens),
|
|
513
|
+
cacheHitRate,
|
|
514
|
+
costSavingsPercent,
|
|
515
|
+
estimatedLatencyReduction,
|
|
516
|
+
providerMetadata: { automatic: true }
|
|
517
|
+
};
|
|
518
|
+
}
|
|
519
|
+
};
|
|
520
|
+
/**
|
|
521
|
+
* No-op adapter for providers without caching support
|
|
522
|
+
*/
|
|
523
|
+
var NoOpCachingAdapter = class {
|
|
524
|
+
applyCaching(apiParams) {
|
|
525
|
+
return apiParams;
|
|
526
|
+
}
|
|
527
|
+
extractCacheStats() {}
|
|
528
|
+
};
|
|
529
|
+
ModelBackend.Anthropic, new AnthropicCachingAdapter(), ModelBackend.OpenAI, new OpenAICachingAdapter(), ModelBackend.Gemini, new GeminiCachingAdapter(), ModelBackend.Bedrock, new AnthropicCachingAdapter(), ModelBackend.XAI, new XAICachingAdapter(), ModelBackend.Kimi, new KimiCachingAdapter(), ModelBackend.DeepSeek, new DeepSeekCachingAdapter(), ModelBackend.Ollama, new NoOpCachingAdapter(), ModelBackend.BFL, new NoOpCachingAdapter(), ModelBackend.VoyageAI, new NoOpCachingAdapter(), ModelBackend.AWS, new NoOpCachingAdapter(), ModelBackend.LocalImage, new NoOpCachingAdapter();
|
|
530
|
+
ChatModels.KIMI_K2_THINKING_BEDROCK, ChatModels.KIMI_K2_5_BEDROCK, ChatModels.DEEPSEEK_R1_BEDROCK, ChatModels.DEEPSEEK_FLASH, ChatModels.DEEPSEEK_V4_PRO;
|
|
531
|
+
new Logger();
|
|
532
|
+
ChatModels.CLAUDE_4_5_SONNET, ChatModels.CLAUDE_4_1_OPUS, ChatModels.CLAUDE_4_5_HAIKU, ChatModels.CLAUDE_4_5_OPUS, ChatModels.CLAUDE_4_6_SONNET, ChatModels.CLAUDE_4_6_OPUS;
|
|
533
|
+
ChatModels.CLAUDE_4_5_SONNET_BEDROCK, ChatModels.CLAUDE_4_5_HAIKU_BEDROCK, ChatModels.CLAUDE_4_5_OPUS_BEDROCK, ChatModels.CLAUDE_4_6_SONNET_BEDROCK, ChatModels.CLAUDE_4_6_OPUS_BEDROCK;
|
|
534
|
+
/** Ollama `done_reason` (chat/generate responses). */
|
|
535
|
+
function normalizeOllamaDoneReason(reason) {
|
|
536
|
+
if (!reason) return void 0;
|
|
537
|
+
switch (reason) {
|
|
538
|
+
case "length": return "max_tokens";
|
|
539
|
+
default: return reason;
|
|
540
|
+
}
|
|
541
|
+
}
|
|
542
|
+
ChatModels.DEEPSEEK_FLASH, ChatModels.DEEPSEEK_V4_PRO;
|
|
543
|
+
/** Type guard: does this message already carry OpenAI-style `tool_calls`? */
|
|
544
|
+
function hasToolCalls(msg) {
|
|
545
|
+
return msg.role === "assistant" && "tool_calls" in msg;
|
|
546
|
+
}
|
|
547
|
+
function isToolUseBlock(block) {
|
|
548
|
+
return block.type === "tool_use";
|
|
549
|
+
}
|
|
550
|
+
function isToolResultBlock(block) {
|
|
551
|
+
return block.type === "tool_result";
|
|
552
|
+
}
|
|
553
|
+
function isTextBlock(block) {
|
|
554
|
+
return block.type === "text";
|
|
555
|
+
}
|
|
556
|
+
/**
|
|
557
|
+
* Convert a single IMessage from B4M standard format to OpenAI-compatible format.
|
|
558
|
+
* Returns an array because a single user message with multiple tool_result blocks
|
|
559
|
+
* expands into multiple OpenAI 'tool' role messages.
|
|
560
|
+
*
|
|
561
|
+
* Messages already in OpenAI format (with `tool_calls` property) pass through unchanged.
|
|
562
|
+
* Messages without tool_use/tool_result content blocks pass through unchanged.
|
|
563
|
+
*/
|
|
564
|
+
function convertMessageToOpenAIFormat(msg, options = {}) {
|
|
565
|
+
if (hasToolCalls(msg)) {
|
|
566
|
+
const reasoningContent = msg.reasoning_content;
|
|
567
|
+
return [{
|
|
568
|
+
role: "assistant",
|
|
569
|
+
content: null,
|
|
570
|
+
tool_calls: msg.tool_calls,
|
|
571
|
+
...options.preserveReasoningContent && typeof reasoningContent === "string" ? { reasoning_content: reasoningContent } : {}
|
|
572
|
+
}];
|
|
573
|
+
}
|
|
574
|
+
if (msg.role === "assistant" && Array.isArray(msg.content)) {
|
|
575
|
+
const contentBlocks = msg.content;
|
|
576
|
+
const toolUseBlocks = contentBlocks.filter(isToolUseBlock);
|
|
577
|
+
if (toolUseBlocks.length > 0) {
|
|
578
|
+
const textParts = contentBlocks.filter(isTextBlock).map((block) => block.text).filter(Boolean);
|
|
579
|
+
return [{
|
|
580
|
+
role: "assistant",
|
|
581
|
+
content: textParts.length > 0 ? textParts.join("\n") : null,
|
|
582
|
+
tool_calls: toolUseBlocks.map((block) => ({
|
|
583
|
+
id: block.id,
|
|
584
|
+
type: "function",
|
|
585
|
+
function: {
|
|
586
|
+
name: block.name,
|
|
587
|
+
arguments: JSON.stringify(block.input)
|
|
588
|
+
}
|
|
589
|
+
}))
|
|
590
|
+
}];
|
|
591
|
+
}
|
|
592
|
+
}
|
|
593
|
+
if (msg.role === "user" && Array.isArray(msg.content)) {
|
|
594
|
+
const toolResultBlocks = msg.content.filter(isToolResultBlock);
|
|
595
|
+
if (toolResultBlocks.length > 0) return toolResultBlocks.map((block) => ({
|
|
596
|
+
role: "tool",
|
|
597
|
+
content: block.content,
|
|
598
|
+
tool_call_id: block.tool_use_id
|
|
599
|
+
}));
|
|
600
|
+
}
|
|
601
|
+
if (msg.requiresTool !== void 0) {
|
|
602
|
+
const { requiresTool: _requiresTool, ...providerSafe } = msg;
|
|
603
|
+
return [providerSafe];
|
|
604
|
+
}
|
|
605
|
+
return [msg];
|
|
606
|
+
}
|
|
607
|
+
/**
|
|
608
|
+
* Convert an array of IMessages from B4M standard format to OpenAI-compatible format.
|
|
609
|
+
* Returns OpenAIFormattedMessage[] - callers targeting OpenAI SDK types should cast at the boundary.
|
|
610
|
+
*/
|
|
611
|
+
function convertMessagesToOpenAIFormat(messages, options = {}) {
|
|
612
|
+
return messages.flatMap((msg) => convertMessageToOpenAIFormat(msg, options));
|
|
613
|
+
}
|
|
614
|
+
ChatModels.KIMI_K3;
|
|
615
|
+
ChatModels.KIMI_K2_7_CODE, ChatModels.KIMI_K2_7_CODE_HIGHSPEED, ChatModels.KIMI_K2_6, ChatModels.KIMI_K2_5;
|
|
616
|
+
ChatModels.KIMI_K2_7_CODE, ChatModels.KIMI_K2_7_CODE_HIGHSPEED;
|
|
617
|
+
ChatModels.KIMI_K2_7_CODE, ChatModels.KIMI_K2_7_CODE_HIGHSPEED, ChatModels.KIMI_K2_6;
|
|
618
|
+
ChatModels.KIMI_K3, ChatModels.KIMI_K2_7_CODE, ChatModels.KIMI_K2_7_CODE_HIGHSPEED, ChatModels.KIMI_K2_6, ChatModels.KIMI_K2_5;
|
|
619
|
+
var OllamaBackend = class OllamaBackend {
|
|
620
|
+
_host;
|
|
621
|
+
_api;
|
|
622
|
+
_logger;
|
|
623
|
+
_clientHost;
|
|
624
|
+
_clientHeaders;
|
|
625
|
+
_agent;
|
|
626
|
+
currentModel = "";
|
|
627
|
+
constructor(host, logger) {
|
|
628
|
+
this._logger = logger ?? new Logger();
|
|
629
|
+
this._host = host ?? "http://localhost:11434";
|
|
630
|
+
const url = new URL(this._host);
|
|
631
|
+
const headers = {};
|
|
632
|
+
if (url.username && url.password) {
|
|
633
|
+
headers.Authorization = `Basic ${Buffer.from(`${url.username}:${url.password}`).toString("base64")}`;
|
|
634
|
+
url.username = "";
|
|
635
|
+
url.password = "";
|
|
636
|
+
}
|
|
637
|
+
this._agent = new Agent({
|
|
638
|
+
headersTimeout: 18e5,
|
|
639
|
+
bodyTimeout: 36e5
|
|
640
|
+
});
|
|
641
|
+
this._clientHost = url.toString();
|
|
642
|
+
this._clientHeaders = headers;
|
|
643
|
+
this._api = this.createClient();
|
|
644
|
+
}
|
|
645
|
+
/**
|
|
646
|
+
* Build an Ollama client. The `ollama` package takes no per-request
|
|
647
|
+
* AbortSignal (it only attaches an internal controller to streaming requests
|
|
648
|
+
* and exposes a client-wide abort()), so cancellation has to be bound into
|
|
649
|
+
* the fetch of a client made for that one request - hence the optional
|
|
650
|
+
* `signal`. The undici Agent is shared across clients so connection pooling
|
|
651
|
+
* and the raised timeouts survive.
|
|
652
|
+
*/
|
|
653
|
+
createClient(signal) {
|
|
654
|
+
const fetchWithTimeout = (input, init) => globalThis.fetch(input, {
|
|
655
|
+
...init,
|
|
656
|
+
dispatcher: this._agent,
|
|
657
|
+
...signal && { signal: init?.signal ? AbortSignal.any([init.signal, signal]) : signal }
|
|
658
|
+
});
|
|
659
|
+
return new Ollama({
|
|
660
|
+
host: this._clientHost,
|
|
661
|
+
headers: this._clientHeaders,
|
|
662
|
+
fetch: fetchWithTimeout
|
|
663
|
+
});
|
|
664
|
+
}
|
|
665
|
+
async getModelInfo() {
|
|
666
|
+
try {
|
|
667
|
+
const models = await this._api.list();
|
|
668
|
+
const isSelfHost = process.env.B4M_SELF_HOST === "true";
|
|
669
|
+
return await Promise.all(models.models.map(async (model) => {
|
|
670
|
+
const { capabilities, contextWindow: reportedContextWindow } = await this.getModelDetailsCached(model.name);
|
|
671
|
+
const contextWindow = OllamaBackend.effectiveContextWindow(reportedContextWindow);
|
|
672
|
+
return {
|
|
673
|
+
id: model.name,
|
|
674
|
+
type: "text",
|
|
675
|
+
name: model.name,
|
|
676
|
+
backend: ModelBackend.Ollama,
|
|
677
|
+
contextWindow,
|
|
678
|
+
max_tokens: OllamaBackend.advertisedOutputCap(contextWindow),
|
|
679
|
+
supportsImageVariation: false,
|
|
680
|
+
pricing: { [contextWindow]: {
|
|
681
|
+
input: 0,
|
|
682
|
+
output: 0
|
|
683
|
+
} },
|
|
684
|
+
freeToRun: true,
|
|
685
|
+
supportsVision: capabilities.includes("vision"),
|
|
686
|
+
supportsTools: capabilities.includes("tools"),
|
|
687
|
+
can_think: capabilities.includes("thinking"),
|
|
688
|
+
can_stream: true,
|
|
689
|
+
logoFile: "Ollama_Logo.svg",
|
|
690
|
+
rank: 1,
|
|
691
|
+
description: isSelfHost ? "Runs locally on your own hardware via Ollama. No API key required, and nothing leaves your machine. Performance and capabilities vary by model." : `This model is served from ${process.env.APP_NAME ? `${process.env.APP_NAME}'s` : "the platform"} Ollama servers using publicly available open-source models. Performance and capabilities vary by model.`
|
|
692
|
+
};
|
|
693
|
+
}));
|
|
694
|
+
} catch (error) {
|
|
695
|
+
let errorMessage = error instanceof Error ? error.message : String(error);
|
|
696
|
+
if (error instanceof Error && error.message.includes("503 Service Temporarily Unavailable")) errorMessage = "Ollama server is temporarily unavailable. Please try again later.";
|
|
697
|
+
this._logger.warn("[OllamaBackend] Error fetching model info from Ollama:", errorMessage);
|
|
698
|
+
return [];
|
|
699
|
+
}
|
|
700
|
+
}
|
|
701
|
+
/** Ollama's default when a model doesn't report a context length. */
|
|
702
|
+
static DEFAULT_CONTEXT_WINDOW = 8192;
|
|
703
|
+
/**
|
|
704
|
+
* Ceiling on the num_ctx we ask Ollama to allocate. Models advertise windows
|
|
705
|
+
* far larger than a dev box can hold in KV cache (262K on qwen3.5), so the
|
|
706
|
+
* useful value is the model's window capped at something affordable rather
|
|
707
|
+
* than the advertised maximum. Override with OLLAMA_MAX_NUM_CTX.
|
|
708
|
+
*/
|
|
709
|
+
static DEFAULT_MAX_NUM_CTX = 32768;
|
|
710
|
+
/**
|
|
711
|
+
* Output ceiling we advertise for a local model. Half of what we allocate, because advertising
|
|
712
|
+
* the whole window made the server's context - output - buffer negative and emptied the prompt.
|
|
713
|
+
*
|
|
714
|
+
* Halving alone is not enough at the bottom of the range: OLLAMA_MAX_NUM_CTX accepts any positive
|
|
715
|
+
* value, so an operator setting 2000 would leave exactly zero input budget for every local model
|
|
716
|
+
* at once. The second bound keeps input room for any window ABOVE the safety buffer.
|
|
717
|
+
*
|
|
718
|
+
* At or below the buffer nothing here can: a 500-token window yields a cap of 1 and still leaves
|
|
719
|
+
* 500 - 1 - 1000 negative. Clamping the window up to hide that would advertise more context than
|
|
720
|
+
* the model has, which is the misreporting this whole change removes, so the honest answer is
|
|
721
|
+
* that such a model cannot serve a chat prompt at all.
|
|
722
|
+
*/
|
|
723
|
+
static advertisedOutputCap(contextWindow) {
|
|
724
|
+
const halved = Math.floor(contextWindow / 2);
|
|
725
|
+
return Math.max(1, Math.min(halved, contextWindow - CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS - 1));
|
|
726
|
+
}
|
|
727
|
+
/** A model's reported window clamped to what we are willing to allocate. */
|
|
728
|
+
static effectiveContextWindow(reported) {
|
|
729
|
+
const configuredCap = Number(process.env.OLLAMA_MAX_NUM_CTX);
|
|
730
|
+
const cap = Number.isFinite(configuredCap) && configuredCap > 0 ? configuredCap : OllamaBackend.DEFAULT_MAX_NUM_CTX;
|
|
731
|
+
return Math.min(reported, cap);
|
|
732
|
+
}
|
|
733
|
+
/** Per-model /api/show results, so the tool-call recursion shows once, not per round. */
|
|
734
|
+
_modelDetails = /* @__PURE__ */ new Map();
|
|
735
|
+
getModelDetailsCached(model) {
|
|
736
|
+
let details = this._modelDetails.get(model);
|
|
737
|
+
if (!details) {
|
|
738
|
+
details = this.getModelDetails(model);
|
|
739
|
+
this._modelDetails.set(model, details);
|
|
740
|
+
}
|
|
741
|
+
return details;
|
|
742
|
+
}
|
|
743
|
+
/**
|
|
744
|
+
* Fetch a model's capabilities and context window from Ollama (/api/show).
|
|
745
|
+
* capabilities is e.g. ['completion', 'tools', 'vision']; the context length
|
|
746
|
+
* lives in model_info under "<architecture>.context_length" (e.g.
|
|
747
|
+
* "qwen2.context_length"). Returns safe defaults on any error so a transient
|
|
748
|
+
* show() failure degrades gracefully instead of dropping the whole list.
|
|
749
|
+
*/
|
|
750
|
+
async getModelDetails(model) {
|
|
751
|
+
try {
|
|
752
|
+
const info = await this._api.show({ model });
|
|
753
|
+
const capabilities = info.capabilities ?? [];
|
|
754
|
+
const raw = info.model_info;
|
|
755
|
+
const ctx = (raw instanceof Map ? Array.from(raw.entries()) : Object.entries(raw ?? {})).find(([k]) => k.endsWith(".context_length"))?.[1];
|
|
756
|
+
return {
|
|
757
|
+
capabilities,
|
|
758
|
+
contextWindow: typeof ctx === "number" && ctx > 0 ? ctx : OllamaBackend.DEFAULT_CONTEXT_WINDOW
|
|
759
|
+
};
|
|
760
|
+
} catch (error) {
|
|
761
|
+
this._logger.debug(`[OllamaBackend] Could not fetch details for ${model}:`, error);
|
|
762
|
+
return {
|
|
763
|
+
capabilities: [],
|
|
764
|
+
contextWindow: OllamaBackend.DEFAULT_CONTEXT_WINDOW
|
|
765
|
+
};
|
|
766
|
+
}
|
|
767
|
+
}
|
|
768
|
+
/**
|
|
769
|
+
* Ollama model parameters for a request. Everything here is silently dropped
|
|
770
|
+
* if the `options` object is omitted, which is why num_ctx matters most:
|
|
771
|
+
* without it the server falls back to its own 4096 default and truncates the
|
|
772
|
+
* prompt from the front, taking the tool block with it - so a large turn
|
|
773
|
+
* makes a tool-capable model answer that it has no tools. Sizing from the
|
|
774
|
+
* model's own reported window keeps that consistent with the context length
|
|
775
|
+
* the picker advertises.
|
|
776
|
+
*/
|
|
777
|
+
async buildModelOptions(model, options) {
|
|
778
|
+
const { contextWindow } = await this.getModelDetailsCached(model);
|
|
779
|
+
const numCtx = OllamaBackend.effectiveContextWindow(contextWindow);
|
|
780
|
+
return {
|
|
781
|
+
num_ctx: numCtx,
|
|
782
|
+
...typeof options.temperature === "number" && { temperature: options.temperature },
|
|
783
|
+
...typeof options.maxTokens === "number" && { num_predict: Math.min(options.maxTokens, numCtx) }
|
|
784
|
+
};
|
|
785
|
+
}
|
|
786
|
+
async complete(model, messages, options, callback) {
|
|
787
|
+
this.currentModel = model;
|
|
788
|
+
const toolCallCount = options._internal?.toolCallCount ?? 0;
|
|
789
|
+
const maxToolCalls = options._internal?.maxToolCalls ?? 10;
|
|
790
|
+
const priorToolsUsed = options._internal?.accumToolsUsed ?? [];
|
|
791
|
+
const priorInputTokens = options._internal?.accumInputTokens ?? 0;
|
|
792
|
+
const priorOutputTokens = options._internal?.accumOutputTokens ?? 0;
|
|
793
|
+
const toolsAvailable = (options.tools?.length ?? 0) > 0;
|
|
794
|
+
const offerTools = toolsAvailable && toolCallCount < maxToolCalls;
|
|
795
|
+
if (toolsAvailable && !offerTools) this._logger.warn(`[OllamaBackend] Max tool calls (${maxToolCalls}) reached; answering without tools.`);
|
|
796
|
+
const formattedTools = offerTools ? this.formatTools(options.tools ?? []) : [];
|
|
797
|
+
const baseRequest = {
|
|
798
|
+
model,
|
|
799
|
+
messages: this.buildMessages(offerTools ? messages : stripToolDependentMessages(messages)),
|
|
800
|
+
options: await this.buildModelOptions(model, options),
|
|
801
|
+
...formattedTools.length > 0 && { tools: formattedTools },
|
|
802
|
+
...typeof options.thinking?.enabled === "boolean" && { think: options.thinking.enabled }
|
|
803
|
+
};
|
|
804
|
+
try {
|
|
805
|
+
const round = await this.runChatRound(baseRequest, options, callback, { buffer: offerTools });
|
|
806
|
+
const inputTokens = priorInputTokens + (round.completionInfo.inputTokens ?? 0);
|
|
807
|
+
const outputTokens = priorOutputTokens + (round.completionInfo.outputTokens ?? 0);
|
|
808
|
+
let toolCalls = this.normalizeToolCalls(round.toolCalls);
|
|
809
|
+
if (toolCalls.length === 0 && offerTools) toolCalls = this.parseContentToolCall(round.content, options.tools ?? []);
|
|
810
|
+
const toolsUsed = [...priorToolsUsed, ...toolCalls.map((tc) => ({
|
|
811
|
+
name: tc.name,
|
|
812
|
+
arguments: tc.arguments,
|
|
813
|
+
id: tc.id
|
|
814
|
+
}))];
|
|
815
|
+
if (toolCalls.length === 0) {
|
|
816
|
+
await callback([offerTools ? round.content : ""], {
|
|
817
|
+
inputTokens,
|
|
818
|
+
outputTokens,
|
|
819
|
+
...toolsUsed.length > 0 && { toolsUsed },
|
|
820
|
+
...round.completionInfo.stopReason ? { stopReason: round.completionInfo.stopReason } : {}
|
|
821
|
+
});
|
|
822
|
+
return;
|
|
823
|
+
}
|
|
824
|
+
if (options.executeTools === false) {
|
|
825
|
+
await callback([""], {
|
|
826
|
+
inputTokens,
|
|
827
|
+
outputTokens,
|
|
828
|
+
toolsUsed
|
|
829
|
+
});
|
|
830
|
+
return;
|
|
831
|
+
}
|
|
832
|
+
const resolved = toolCalls.map((tc) => ({
|
|
833
|
+
tc,
|
|
834
|
+
toolFn: options.tools?.find((t) => t.toolSchema.name === tc.name)?.toolFn
|
|
835
|
+
})).filter((r) => !!r.toolFn);
|
|
836
|
+
const unknownCalls = toolCalls.filter((tc) => !options.tools?.some((t) => t.toolSchema.name === tc.name));
|
|
837
|
+
for (const tc of unknownCalls) this.pushToolMessages(messages, {
|
|
838
|
+
id: tc.id,
|
|
839
|
+
name: tc.name,
|
|
840
|
+
parameters: tc.arguments || "{}"
|
|
841
|
+
}, `Error: tool "${tc.name}" is not available. Do not call it again; answer directly or use a listed tool.`);
|
|
842
|
+
const outcomes = await executeToolsBatch(resolved.map(({ tc, toolFn }) => async () => {
|
|
843
|
+
let params = {};
|
|
844
|
+
try {
|
|
845
|
+
params = JSON.parse(tc.arguments || "{}");
|
|
846
|
+
} catch {}
|
|
847
|
+
this._logger.debug(`[OllamaBackend] Executing tool ${tc.name}`);
|
|
848
|
+
return String(await toolFn(params));
|
|
849
|
+
}), {
|
|
850
|
+
parallel: options.parallelToolExecution !== false,
|
|
851
|
+
maxConcurrency: options.maxParallelTools
|
|
852
|
+
});
|
|
853
|
+
const observations = [];
|
|
854
|
+
outcomes.forEach((outcome, i) => {
|
|
855
|
+
const { tc } = resolved[i];
|
|
856
|
+
const params = tc.arguments || "{}";
|
|
857
|
+
if (outcome.ok) {
|
|
858
|
+
observations[i] = outcome.result;
|
|
859
|
+
this.pushToolMessages(messages, {
|
|
860
|
+
id: tc.id,
|
|
861
|
+
name: tc.name,
|
|
862
|
+
parameters: params
|
|
863
|
+
}, outcome.result);
|
|
864
|
+
} else {
|
|
865
|
+
if (outcome.error instanceof PermissionDeniedError) throw outcome.error;
|
|
866
|
+
const errorMsg = `Error running ${tc.name}: ${outcome.error instanceof Error ? outcome.error.message : "Unknown error"}`;
|
|
867
|
+
observations[i] = errorMsg;
|
|
868
|
+
this.pushToolMessages(messages, {
|
|
869
|
+
id: tc.id,
|
|
870
|
+
name: tc.name,
|
|
871
|
+
parameters: params
|
|
872
|
+
}, errorMsg);
|
|
873
|
+
}
|
|
874
|
+
});
|
|
875
|
+
const executedToolsUsed = [...priorToolsUsed, ...resolved.map(({ tc }, i) => {
|
|
876
|
+
const entry = {
|
|
877
|
+
name: tc.name,
|
|
878
|
+
arguments: tc.arguments,
|
|
879
|
+
id: tc.id,
|
|
880
|
+
returnValue: truncateToolResult(String(observations[i])),
|
|
881
|
+
success: outcomes[i].ok
|
|
882
|
+
};
|
|
883
|
+
attachFullToolResult(entry, observations[i]);
|
|
884
|
+
return entry;
|
|
885
|
+
})];
|
|
886
|
+
if (options.abortSignal?.aborted) {
|
|
887
|
+
await callback([""], {
|
|
888
|
+
inputTokens,
|
|
889
|
+
outputTokens,
|
|
890
|
+
toolsUsed: executedToolsUsed
|
|
891
|
+
});
|
|
892
|
+
return;
|
|
893
|
+
}
|
|
894
|
+
await this.complete(model, messages, {
|
|
895
|
+
...options,
|
|
896
|
+
_internal: {
|
|
897
|
+
...options._internal,
|
|
898
|
+
toolCallCount: toolCallCount + 1,
|
|
899
|
+
accumToolsUsed: executedToolsUsed,
|
|
900
|
+
accumInputTokens: inputTokens,
|
|
901
|
+
accumOutputTokens: outputTokens
|
|
902
|
+
}
|
|
903
|
+
}, callback);
|
|
904
|
+
} catch (error) {
|
|
905
|
+
if (error instanceof Error && isUserInitiatedAbort(error, options.abortSignal)) this._logger.debug("[OllamaBackend] Ollama request cancelled by the caller");
|
|
906
|
+
else this._logger.error("[OllamaBackend] Error during Ollama API call:", error);
|
|
907
|
+
throw error;
|
|
908
|
+
}
|
|
909
|
+
}
|
|
910
|
+
/**
|
|
911
|
+
* Run a single Ollama chat turn. Streams text chunks to `callback` unless
|
|
912
|
+
* `buffer` is set (used for tool-eligible rounds, where the content is
|
|
913
|
+
* withheld until we know whether it is a tool call or the final answer).
|
|
914
|
+
* Returns the full text, any native tool calls, and token usage.
|
|
915
|
+
*/
|
|
916
|
+
async runChatRound(baseRequest, options, callback, { buffer }) {
|
|
917
|
+
const toolCalls = [];
|
|
918
|
+
let content = "";
|
|
919
|
+
let inputTokens = 0;
|
|
920
|
+
let outputTokens = 0;
|
|
921
|
+
let doneReason;
|
|
922
|
+
const api = options.abortSignal ? this.createClient(options.abortSignal) : this._api;
|
|
923
|
+
if (options.stream) {
|
|
924
|
+
const response = await api.chat({
|
|
925
|
+
...baseRequest,
|
|
926
|
+
stream: true
|
|
927
|
+
});
|
|
928
|
+
let startedThinking = false;
|
|
929
|
+
let stoppedThinking = false;
|
|
930
|
+
let thinkingFieldOpen = false;
|
|
931
|
+
const thinkEscaper = createThinkMarkerEscaper();
|
|
932
|
+
for await (const chunk of response) {
|
|
933
|
+
if (chunk.message.tool_calls?.length) toolCalls.push(...chunk.message.tool_calls);
|
|
934
|
+
let piece = "";
|
|
935
|
+
const thinkPiece = chunk.message.thinking || "";
|
|
936
|
+
if (thinkPiece) {
|
|
937
|
+
if (!thinkingFieldOpen) {
|
|
938
|
+
piece += "<think>";
|
|
939
|
+
thinkingFieldOpen = true;
|
|
940
|
+
}
|
|
941
|
+
piece += thinkEscaper.push(thinkPiece);
|
|
942
|
+
}
|
|
943
|
+
const contentPiece = chunk.message.content || "";
|
|
944
|
+
if (contentPiece) {
|
|
945
|
+
if (thinkingFieldOpen) {
|
|
946
|
+
piece += thinkEscaper.flush();
|
|
947
|
+
piece += "</think>";
|
|
948
|
+
thinkingFieldOpen = false;
|
|
949
|
+
}
|
|
950
|
+
startedThinking = startedThinking || contentPiece.includes("<think>");
|
|
951
|
+
stoppedThinking = stoppedThinking || contentPiece.includes("</think>");
|
|
952
|
+
piece += contentPiece;
|
|
953
|
+
}
|
|
954
|
+
if (chunk.done && (thinkingFieldOpen || startedThinking && !stoppedThinking)) {
|
|
955
|
+
if (thinkingFieldOpen) piece += thinkEscaper.flush();
|
|
956
|
+
piece = `${piece}</think>`;
|
|
957
|
+
thinkingFieldOpen = false;
|
|
958
|
+
}
|
|
959
|
+
content += piece;
|
|
960
|
+
inputTokens = Math.max(inputTokens, chunk.prompt_eval_count || 0);
|
|
961
|
+
outputTokens += chunk.eval_count || 0;
|
|
962
|
+
if (chunk.done_reason) doneReason = chunk.done_reason;
|
|
963
|
+
if (!buffer && piece) await callback([piece], {
|
|
964
|
+
inputTokens,
|
|
965
|
+
outputTokens
|
|
966
|
+
});
|
|
967
|
+
}
|
|
968
|
+
} else {
|
|
969
|
+
const response = await api.chat({
|
|
970
|
+
...baseRequest,
|
|
971
|
+
stream: false
|
|
972
|
+
});
|
|
973
|
+
if (response.message.tool_calls?.length) toolCalls.push(...response.message.tool_calls);
|
|
974
|
+
const think = response.message.thinking || "";
|
|
975
|
+
content = (think ? `<think>${escapeThinkMarkers(think)}</think>` : "") + (response.message.content || "");
|
|
976
|
+
inputTokens = response.prompt_eval_count || 0;
|
|
977
|
+
outputTokens = response.eval_count || 0;
|
|
978
|
+
doneReason = response.done_reason;
|
|
979
|
+
if (!buffer) await callback([content], {
|
|
980
|
+
inputTokens,
|
|
981
|
+
outputTokens
|
|
982
|
+
});
|
|
983
|
+
}
|
|
984
|
+
const stopReason = normalizeOllamaDoneReason(doneReason);
|
|
985
|
+
return {
|
|
986
|
+
content,
|
|
987
|
+
toolCalls,
|
|
988
|
+
completionInfo: {
|
|
989
|
+
inputTokens,
|
|
990
|
+
outputTokens,
|
|
991
|
+
...stopReason ? { stopReason } : {}
|
|
992
|
+
}
|
|
993
|
+
};
|
|
994
|
+
}
|
|
995
|
+
/**
|
|
996
|
+
* Normalize Ollama's native tool_calls into the shared NormalizedToolCall shape.
|
|
997
|
+
*
|
|
998
|
+
* Ids are real uuids, not a position-derived string. A prior version keyed ids off
|
|
999
|
+
* `accumulated-count + round-local-index`, but the accumulated count is measured AFTER
|
|
1000
|
+
* hallucinated calls are filtered out while the round-local index is assigned BEFORE that
|
|
1001
|
+
* filter runs, so the two can drift and mint the same id for two different real calls across
|
|
1002
|
+
* rounds - replayableToolCalls dedupes by id and silently drops the later one. A uuid makes
|
|
1003
|
+
* the whole collision class unrepresentable, matching how the other backends already mint ids.
|
|
1004
|
+
*/
|
|
1005
|
+
normalizeToolCalls(toolCalls) {
|
|
1006
|
+
return toolCalls.map((tc) => ({
|
|
1007
|
+
name: tc.function.name,
|
|
1008
|
+
arguments: JSON.stringify(tc.function.arguments ?? {}),
|
|
1009
|
+
id: `ollama-tool-${v4()}`
|
|
1010
|
+
}));
|
|
1011
|
+
}
|
|
1012
|
+
/**
|
|
1013
|
+
* Some smaller models emit tool calls as plain message content instead of
|
|
1014
|
+
* using the native tool_calls field: a bare {"name":...,"arguments":{...}},
|
|
1015
|
+
* the same wrapped in a ```json fence, or several such objects run together
|
|
1016
|
+
* ({...} {...}). Recover every such object that names an available tool; if
|
|
1017
|
+
* none match, the content is a normal answer.
|
|
1018
|
+
*
|
|
1019
|
+
* Guards against false positives: reasoning traces (<think>...</think>) are
|
|
1020
|
+
* stripped first. The search source is a leading bare object (only when the
|
|
1021
|
+
* reply STARTS with one) plus every fenced body - so a call wrapped in a fence
|
|
1022
|
+
* after preamble prose is recovered, a bare call is not lost just because the
|
|
1023
|
+
* reply also contains an unrelated fence, and JSON merely quoted mid-prose
|
|
1024
|
+
* ("the math_evaluate tool takes {...}") is ignored. The seen-set dedupes any
|
|
1025
|
+
* overlap between the two sources.
|
|
1026
|
+
*
|
|
1027
|
+
* Because any fenced block is a search source, a fenced EXAMPLE of a real
|
|
1028
|
+
* (authorized) tool is also recovered and executed. This stays bounded to
|
|
1029
|
+
* authorized tools (tryParseToolCallJson requires a name present in `tools`),
|
|
1030
|
+
* so there is no privilege escalation - only a wider trigger surface.
|
|
1031
|
+
*/
|
|
1032
|
+
parseContentToolCall(content, tools) {
|
|
1033
|
+
const withoutThink = content.replace(/<think>[\s\S]*?<\/think>/gi, "").trim();
|
|
1034
|
+
const fenced = this.extractFencedBlocks(withoutThink);
|
|
1035
|
+
const source = [withoutThink.startsWith("{") ? withoutThink : "", ...fenced].join("\n");
|
|
1036
|
+
if (!source.trim()) return [];
|
|
1037
|
+
const calls = [];
|
|
1038
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1039
|
+
for (const candidate of this.extractJsonObjects(source)) {
|
|
1040
|
+
const call = this.tryParseToolCallJson(candidate, tools);
|
|
1041
|
+
if (!call) continue;
|
|
1042
|
+
const key = `${call.name}:${call.arguments}`;
|
|
1043
|
+
if (seen.has(key)) continue;
|
|
1044
|
+
seen.add(key);
|
|
1045
|
+
calls.push({
|
|
1046
|
+
...call,
|
|
1047
|
+
id: `ollama-content-tool-${v4()}`
|
|
1048
|
+
});
|
|
1049
|
+
}
|
|
1050
|
+
return calls;
|
|
1051
|
+
}
|
|
1052
|
+
/**
|
|
1053
|
+
* Return the body of every ```-fenced block, so a tool call the model wrapped
|
|
1054
|
+
* in a fence after preamble prose can be isolated from the surrounding text.
|
|
1055
|
+
*/
|
|
1056
|
+
extractFencedBlocks(content) {
|
|
1057
|
+
const blocks = [];
|
|
1058
|
+
const fenceRegex = /```[^\n]*\n?([\s\S]*?)```/g;
|
|
1059
|
+
let match;
|
|
1060
|
+
while ((match = fenceRegex.exec(content)) !== null) blocks.push(match[1]);
|
|
1061
|
+
return blocks;
|
|
1062
|
+
}
|
|
1063
|
+
/**
|
|
1064
|
+
* Extract every balanced top-level {...} substring from arbitrary text. String
|
|
1065
|
+
* contents are respected so braces inside JSON strings don't throw off nesting,
|
|
1066
|
+
* and code fences / prose around the objects are ignored. Handles multiple
|
|
1067
|
+
* objects run together, which is how some models emit parallel tool calls.
|
|
1068
|
+
*/
|
|
1069
|
+
extractJsonObjects(content) {
|
|
1070
|
+
const objects = [];
|
|
1071
|
+
let depth = 0;
|
|
1072
|
+
let start = -1;
|
|
1073
|
+
let inString = false;
|
|
1074
|
+
let escaped = false;
|
|
1075
|
+
for (let i = 0; i < content.length; i++) {
|
|
1076
|
+
const ch = content[i];
|
|
1077
|
+
if (inString) {
|
|
1078
|
+
if (escaped) escaped = false;
|
|
1079
|
+
else if (ch === "\\") escaped = true;
|
|
1080
|
+
else if (ch === "\"") inString = false;
|
|
1081
|
+
continue;
|
|
1082
|
+
}
|
|
1083
|
+
if (ch === "\"") inString = true;
|
|
1084
|
+
else if (ch === "{") {
|
|
1085
|
+
if (depth === 0) start = i;
|
|
1086
|
+
depth++;
|
|
1087
|
+
} else if (ch === "}" && depth > 0) {
|
|
1088
|
+
depth--;
|
|
1089
|
+
if (depth === 0 && start !== -1) {
|
|
1090
|
+
objects.push(content.slice(start, i + 1));
|
|
1091
|
+
start = -1;
|
|
1092
|
+
}
|
|
1093
|
+
}
|
|
1094
|
+
}
|
|
1095
|
+
return objects;
|
|
1096
|
+
}
|
|
1097
|
+
/**
|
|
1098
|
+
* Parse one candidate string as a tool call naming a known tool. Models
|
|
1099
|
+
* improvise the shape, so accept the common ones:
|
|
1100
|
+
* {"name":"t","arguments":{...}} (Ollama/most)
|
|
1101
|
+
* {"function":"t","arguments":{...}} (function-as-name)
|
|
1102
|
+
* {"function":{"name":"t","arguments":{}}} (OpenAI-style nested)
|
|
1103
|
+
* plus "parameters"/"args" aliases for the arguments.
|
|
1104
|
+
*/
|
|
1105
|
+
tryParseToolCallJson(text, tools) {
|
|
1106
|
+
const trimmed = text.trim();
|
|
1107
|
+
if (!trimmed.startsWith("{") || !trimmed.endsWith("}")) return null;
|
|
1108
|
+
let parsed;
|
|
1109
|
+
try {
|
|
1110
|
+
parsed = JSON.parse(trimmed);
|
|
1111
|
+
} catch {
|
|
1112
|
+
return null;
|
|
1113
|
+
}
|
|
1114
|
+
if (!parsed || typeof parsed !== "object") return null;
|
|
1115
|
+
const obj = parsed;
|
|
1116
|
+
let name;
|
|
1117
|
+
let args;
|
|
1118
|
+
const fn = obj.function;
|
|
1119
|
+
if (fn && typeof fn === "object") {
|
|
1120
|
+
name = fn.name;
|
|
1121
|
+
args = fn.arguments;
|
|
1122
|
+
} else name = obj.name ?? obj.function ?? obj.tool ?? obj.tool_name;
|
|
1123
|
+
if (args === void 0) args = obj.arguments ?? obj.parameters ?? obj.args ?? {};
|
|
1124
|
+
if (typeof name !== "string" || !tools.some((t) => t.toolSchema.name === name)) return null;
|
|
1125
|
+
return {
|
|
1126
|
+
name,
|
|
1127
|
+
arguments: typeof args === "string" ? args : JSON.stringify(args ?? {})
|
|
1128
|
+
};
|
|
1129
|
+
}
|
|
1130
|
+
pushToolMessages(messages, tool, result, _thinkingBlocks) {
|
|
1131
|
+
let argumentsObj;
|
|
1132
|
+
try {
|
|
1133
|
+
argumentsObj = JSON.parse(tool.parameters);
|
|
1134
|
+
} catch {
|
|
1135
|
+
argumentsObj = { _raw: tool.parameters };
|
|
1136
|
+
}
|
|
1137
|
+
messages.push({
|
|
1138
|
+
content: "",
|
|
1139
|
+
role: "assistant",
|
|
1140
|
+
tool_calls: [{ function: {
|
|
1141
|
+
name: tool.name,
|
|
1142
|
+
arguments: argumentsObj
|
|
1143
|
+
} }]
|
|
1144
|
+
});
|
|
1145
|
+
messages.push({
|
|
1146
|
+
role: "tool",
|
|
1147
|
+
tool_name: tool.name,
|
|
1148
|
+
content: result
|
|
1149
|
+
});
|
|
1150
|
+
}
|
|
1151
|
+
/**
|
|
1152
|
+
* Convert ICompletionOptionTools into Ollama's Tool schema format.
|
|
1153
|
+
*/
|
|
1154
|
+
formatTools(tools) {
|
|
1155
|
+
return tools.map((tool) => ({
|
|
1156
|
+
type: "function",
|
|
1157
|
+
function: {
|
|
1158
|
+
...tool.toolSchema,
|
|
1159
|
+
parameters: {
|
|
1160
|
+
...tool.toolSchema.parameters,
|
|
1161
|
+
required: tool.toolSchema.parameters.required ?? []
|
|
1162
|
+
}
|
|
1163
|
+
}
|
|
1164
|
+
}));
|
|
1165
|
+
}
|
|
1166
|
+
/**
|
|
1167
|
+
* Map IMessage[] to Ollama's Message[], preserving tool_calls for multi-turn
|
|
1168
|
+
* tool conversations (added by pushToolMessages).
|
|
1169
|
+
* First converts B4M standard format (tool_use/tool_result) to OpenAI-compatible
|
|
1170
|
+
* format since Ollama uses the same tool_calls/role:tool convention.
|
|
1171
|
+
*/
|
|
1172
|
+
buildMessages(messages) {
|
|
1173
|
+
return convertMessagesToOpenAIFormat(messages).map((msg) => {
|
|
1174
|
+
const raw = msg;
|
|
1175
|
+
const mapped = {
|
|
1176
|
+
role: msg.role,
|
|
1177
|
+
content: ""
|
|
1178
|
+
};
|
|
1179
|
+
if (Array.isArray(msg.content)) {
|
|
1180
|
+
const texts = [];
|
|
1181
|
+
const images = [];
|
|
1182
|
+
for (const block of msg.content) if (block.type === "text" && typeof block.text === "string") texts.push(block.text);
|
|
1183
|
+
else if (block.type === "image") if (block.source.type === "base64") images.push(block.source.data);
|
|
1184
|
+
else this._logger.debug("[OllamaBackend] Dropping non-base64 image block; Ollama requires inline base64.");
|
|
1185
|
+
else if (block.type === "image_url") {
|
|
1186
|
+
const dataUrl = block.image_url.url.match(/^data:[^,]*;base64,(.+)$/s);
|
|
1187
|
+
if (dataUrl) images.push(dataUrl[1]);
|
|
1188
|
+
else this._logger.debug("[OllamaBackend] Dropping non-data image_url; Ollama requires inline base64.");
|
|
1189
|
+
}
|
|
1190
|
+
mapped.content = texts.join("\n");
|
|
1191
|
+
if (images.length > 0) mapped.images = images;
|
|
1192
|
+
} else mapped.content = msg.content != null ? String(msg.content) : "";
|
|
1193
|
+
if (Array.isArray(raw.tool_calls)) mapped.tool_calls = raw.tool_calls;
|
|
1194
|
+
if (typeof raw.tool_name === "string") mapped.tool_name = raw.tool_name;
|
|
1195
|
+
return mapped;
|
|
1196
|
+
});
|
|
1197
|
+
}
|
|
1198
|
+
async listModels() {
|
|
1199
|
+
try {
|
|
1200
|
+
this._logger.debug("[OllamaBackend] Listing models from Ollama");
|
|
1201
|
+
const response = await this._api.list();
|
|
1202
|
+
this._logger.debug("[OllamaBackend] Models listed from Ollama:", response.models);
|
|
1203
|
+
return response.models;
|
|
1204
|
+
} catch (error) {
|
|
1205
|
+
this._logger.error("[OllamaBackend] Error listing models from Ollama:", error);
|
|
1206
|
+
if (error.message?.includes("ECONNREFUSED") || error.message?.includes("Failed to fetch")) throw new Error(`Could not connect to Ollama. Please make sure it is running at ${this._host}`);
|
|
1207
|
+
throw error;
|
|
1208
|
+
}
|
|
1209
|
+
}
|
|
1210
|
+
};
|
|
1211
|
+
ChatModels.O1_PREVIEW, ChatModels.O1_MINI, ChatModels.O1, ChatModels.O3_MINI, ChatModels.O3, ChatModels.O4_MINI;
|
|
1212
|
+
ChatModels.GPT5, ChatModels.GPT5_MINI, ChatModels.GPT5_NANO, ChatModels.GPT5_CHAT_LATEST;
|
|
1213
|
+
ChatModels.GPT5_1, ChatModels.GPT5_1_CHAT_LATEST;
|
|
1214
|
+
ChatModels.GPT5_2, ChatModels.GPT5_2_CHAT_LATEST;
|
|
1215
|
+
ChatModels.GPT5_4, ChatModels.GPT5_4_MINI, ChatModels.GPT5_4_NANO;
|
|
1216
|
+
ChatModels.GPT5_5;
|
|
1217
|
+
ChatModels.GPT5_6_SOL, ChatModels.GPT5_6_LUNA, ChatModels.GPT5_6_TERRA;
|
|
1218
|
+
ModelBackend.Bedrock, ModelBackend.AWS, ModelBackend.OpenAI, ModelBackend.Anthropic, ModelBackend.Gemini, ModelBackend.Ollama, ModelBackend.BFL, ModelBackend.XAI, ModelBackend.Kimi, ModelBackend.DeepSeek, ModelBackend.LocalImage, ModelBackend.VoyageAI;
|
|
1219
|
+
Object.values(ModelBackend);
|
|
1220
|
+
Object.entries(MODEL_INFO_FIELD_GROUP_OF);
|
|
1221
|
+
ModelBackend.Anthropic, ModelBackend.Gemini, ModelBackend.XAI, ModelBackend.Ollama, ModelBackend.BFL, ModelBackend.LocalImage, ModelBackend.AWS;
|
|
1222
|
+
ModelBackend.Ollama, ModelBackend.LocalImage;
|
|
1223
|
+
//#endregion
|
|
1224
|
+
//#region src/llm/NotifyingLlmBackend.ts
|
|
1225
|
+
/**
|
|
1226
|
+
* LLM backend wrapper that injects background agent notifications
|
|
1227
|
+
* into the message array before each completion call.
|
|
1228
|
+
*
|
|
1229
|
+
* When a background agent completes (or fails), the notification
|
|
1230
|
+
* appears as a system message so the main agent naturally sees it
|
|
1231
|
+
* in context - no polling required.
|
|
1232
|
+
*/
|
|
1233
|
+
var NotifyingLlmBackend = class {
|
|
1234
|
+
constructor(inner, backgroundManager) {
|
|
1235
|
+
this.inner = inner;
|
|
1236
|
+
this.backgroundManager = backgroundManager;
|
|
1237
|
+
}
|
|
1238
|
+
get currentModel() {
|
|
1239
|
+
return this.inner.currentModel;
|
|
1240
|
+
}
|
|
1241
|
+
set currentModel(model) {
|
|
1242
|
+
this.inner.currentModel = model;
|
|
1243
|
+
}
|
|
1244
|
+
async complete(model, messages, options, callback) {
|
|
1245
|
+
const notifications = this.backgroundManager.drainNotifications();
|
|
1246
|
+
let effectiveMessages = messages;
|
|
1247
|
+
if (notifications.length > 0) {
|
|
1248
|
+
const notificationMessage = {
|
|
1249
|
+
role: "user",
|
|
1250
|
+
content: `[System Notification]\n\n${notifications.join("\n\n---\n\n")}\n\nPlease acknowledge these background agent results and incorporate them into your current work.`
|
|
1251
|
+
};
|
|
1252
|
+
effectiveMessages = [...messages, notificationMessage];
|
|
1253
|
+
}
|
|
1254
|
+
return this.inner.complete(model, effectiveMessages, options, callback);
|
|
1255
|
+
}
|
|
1256
|
+
pushToolMessages(messages, tool, result) {
|
|
1257
|
+
return this.inner.pushToolMessages(messages, tool, result);
|
|
1258
|
+
}
|
|
1259
|
+
async getModelInfo() {
|
|
1260
|
+
return this.inner.getModelInfo();
|
|
1261
|
+
}
|
|
1262
|
+
};
|
|
1263
|
+
//#endregion
|
|
1264
|
+
//#region src/tools/deferredToolRegistry.ts
|
|
1265
|
+
/**
|
|
1266
|
+
* Registry of tool schemas that are NOT loaded into the model's initial tool
|
|
1267
|
+
* list. The model sees only the names (via the system prompt directory) and
|
|
1268
|
+
* must call the `tool_search` meta-tool to load schemas on demand.
|
|
1269
|
+
*
|
|
1270
|
+
* This mirrors Claude Code's deferred-tool pattern. The win is large for
|
|
1271
|
+
* heavy MCP integrations (e.g. 41 GitHub MCP tools at ~250-350 tokens of
|
|
1272
|
+
* JSONSchema each = ~10-15k tokens per turn that's now ~1-1.5k of names).
|
|
1273
|
+
*/
|
|
1274
|
+
var DeferredToolRegistry = class {
|
|
1275
|
+
constructor() {
|
|
1276
|
+
this.byName = /* @__PURE__ */ new Map();
|
|
1277
|
+
this.directoryNames = [];
|
|
1278
|
+
}
|
|
1279
|
+
/** Replace registry contents with the supplied tools. Idempotent. */
|
|
1280
|
+
register(tools) {
|
|
1281
|
+
this.byName.clear();
|
|
1282
|
+
for (const tool of tools) this.byName.set(tool.toolSchema.name, tool);
|
|
1283
|
+
this.directoryNames = Object.freeze([...this.byName.keys()].sort());
|
|
1284
|
+
logger.debug(`[DeferredToolRegistry] Registered ${tools.length} deferred tool(s)`);
|
|
1285
|
+
}
|
|
1286
|
+
clear() {
|
|
1287
|
+
this.byName.clear();
|
|
1288
|
+
this.directoryNames = Object.freeze([]);
|
|
1289
|
+
}
|
|
1290
|
+
size() {
|
|
1291
|
+
return this.byName.size;
|
|
1292
|
+
}
|
|
1293
|
+
has(name) {
|
|
1294
|
+
return this.byName.has(name);
|
|
1295
|
+
}
|
|
1296
|
+
get(name) {
|
|
1297
|
+
return this.byName.get(name);
|
|
1298
|
+
}
|
|
1299
|
+
getAll() {
|
|
1300
|
+
return Array.from(this.byName.values());
|
|
1301
|
+
}
|
|
1302
|
+
/** Return tools whose names appear in the supplied list, in input order. */
|
|
1303
|
+
getByNames(names) {
|
|
1304
|
+
const found = [];
|
|
1305
|
+
for (const name of names) {
|
|
1306
|
+
const tool = this.byName.get(name);
|
|
1307
|
+
if (tool) found.push(tool);
|
|
1308
|
+
}
|
|
1309
|
+
return found;
|
|
1310
|
+
}
|
|
1311
|
+
/**
|
|
1312
|
+
* Rank-search deferred tools by query terms. Name matches outrank
|
|
1313
|
+
* description matches; exact substring on name wins ties.
|
|
1314
|
+
*/
|
|
1315
|
+
searchByKeywords(query, maxResults) {
|
|
1316
|
+
const terms = query.toLowerCase().split(/\s+/).filter((t) => t.length > 0);
|
|
1317
|
+
if (terms.length === 0) return [];
|
|
1318
|
+
const scored = [];
|
|
1319
|
+
for (const tool of this.byName.values()) {
|
|
1320
|
+
const name = tool.toolSchema.name.toLowerCase();
|
|
1321
|
+
const desc = (tool.toolSchema.description || "").toLowerCase();
|
|
1322
|
+
let score = 0;
|
|
1323
|
+
for (const term of terms) {
|
|
1324
|
+
if (name.includes(term)) score += 10;
|
|
1325
|
+
if (desc.includes(term)) score += 1;
|
|
1326
|
+
}
|
|
1327
|
+
if (score > 0) scored.push({
|
|
1328
|
+
tool,
|
|
1329
|
+
score
|
|
1330
|
+
});
|
|
1331
|
+
}
|
|
1332
|
+
scored.sort((a, b) => b.score - a.score);
|
|
1333
|
+
return scored.slice(0, maxResults).map((s) => s.tool);
|
|
1334
|
+
}
|
|
1335
|
+
/**
|
|
1336
|
+
* Directory entries rendered into the cache-stamped system-prompt reminder.
|
|
1337
|
+
* Returns the frozen snapshot captured at `register()`, NOT live `byName`
|
|
1338
|
+
* keys, so loading a tool mid-session can never change a byte of the cached
|
|
1339
|
+
* system block (issue #213). A loaded tool remaining listed here is
|
|
1340
|
+
* harmless: re-selecting it via `tool_search` is an idempotent no-op.
|
|
1341
|
+
*/
|
|
1342
|
+
getDirectoryNames() {
|
|
1343
|
+
return [...this.directoryNames];
|
|
1344
|
+
}
|
|
1345
|
+
};
|
|
1346
|
+
const deferredToolRegistry = new DeferredToolRegistry();
|
|
1347
|
+
//#endregion
|
|
1348
|
+
//#region src/tools/toolSearchTool.ts
|
|
1349
|
+
/**
|
|
1350
|
+
* Default number of tools returned for a keyword search. Matches Claude
|
|
1351
|
+
* Code's ToolSearch convention. 5 keeps the response payload small while
|
|
1352
|
+
* surfacing enough alternatives for the model to refine its query.
|
|
1353
|
+
*/
|
|
1354
|
+
const DEFAULT_MAX_RESULTS = 5;
|
|
1355
|
+
/**
|
|
1356
|
+
* Runtime validation for tool_search params. The LLM produces these
|
|
1357
|
+
* values, so we validate at this boundary rather than trusting the
|
|
1358
|
+
* shape. Coerces `max_results` from string->number for models that emit
|
|
1359
|
+
* numeric-looking strings.
|
|
1360
|
+
*/
|
|
1361
|
+
const ToolSearchParamsSchema = z$1.object({
|
|
1362
|
+
query: z$1.string().min(1, "query must be a non-empty string"),
|
|
1363
|
+
max_results: z$1.coerce.number().int().min(1).max(20).optional()
|
|
1364
|
+
});
|
|
1365
|
+
/**
|
|
1366
|
+
* Parse the query string. Two forms:
|
|
1367
|
+
* - `select:name1,name2,...` - exact-name selection
|
|
1368
|
+
* - free text - keyword search across name + description
|
|
1369
|
+
*/
|
|
1370
|
+
function parseQuery(query) {
|
|
1371
|
+
const trimmed = query.trim();
|
|
1372
|
+
const selectMatch = trimmed.match(/^select:(.+)$/i);
|
|
1373
|
+
if (selectMatch) return {
|
|
1374
|
+
mode: "select",
|
|
1375
|
+
names: selectMatch[1].split(",").map((n) => n.trim()).filter((n) => n.length > 0)
|
|
1376
|
+
};
|
|
1377
|
+
return {
|
|
1378
|
+
mode: "search",
|
|
1379
|
+
text: trimmed
|
|
1380
|
+
};
|
|
1381
|
+
}
|
|
1382
|
+
/**
|
|
1383
|
+
* Format the loaded-tools response. Mirrors Claude Code's convention:
|
|
1384
|
+
* one <function>{...}</function> line per matched tool. The model has
|
|
1385
|
+
* already seen this format in its tool-registration system messages, so
|
|
1386
|
+
* it parses without additional explanation.
|
|
1387
|
+
*
|
|
1388
|
+
* Note: the schemas are *also* injected into context.tools by the caller,
|
|
1389
|
+
* so on the next iteration the model gets them as native tool definitions.
|
|
1390
|
+
* The text response here is for in-turn awareness and audit trail.
|
|
1391
|
+
*/
|
|
1392
|
+
function renderToolsBlock(tools) {
|
|
1393
|
+
if (tools.length === 0) return "";
|
|
1394
|
+
return `<functions>\n${tools.map((tool) => {
|
|
1395
|
+
const schema = {
|
|
1396
|
+
description: tool.toolSchema.description,
|
|
1397
|
+
name: tool.toolSchema.name,
|
|
1398
|
+
parameters: tool.toolSchema.parameters
|
|
1399
|
+
};
|
|
1400
|
+
return `<function>${JSON.stringify(schema)}</function>`;
|
|
1401
|
+
}).join("\n")}\n</functions>`;
|
|
1402
|
+
}
|
|
1403
|
+
/**
|
|
1404
|
+
* Build the tool_search meta-tool. The returned tool has a closure over
|
|
1405
|
+
* the supplied `toolListAccessor`, which it uses to push newly-resolved
|
|
1406
|
+
* tool schemas into the live agent context.
|
|
1407
|
+
*
|
|
1408
|
+
* Idempotent: re-loading a tool that's already in the context is a no-op.
|
|
1409
|
+
*/
|
|
1410
|
+
function createToolSearchTool(toolListAccessor) {
|
|
1411
|
+
return {
|
|
1412
|
+
toolSchema: {
|
|
1413
|
+
name: "tool_search",
|
|
1414
|
+
description: "Fetches full schema definitions for deferred tools so they can be called. Deferred tools appear by name only in a system reminder; their parameter schemas are NOT loaded by default. Use this tool to load schemas on demand. Query forms: 'select:name1,name2' for exact selection, or free-text keywords to search by name and description. Once a tool's schema is returned, it becomes callable in subsequent turns.",
|
|
1415
|
+
parameters: {
|
|
1416
|
+
type: "object",
|
|
1417
|
+
properties: {
|
|
1418
|
+
query: {
|
|
1419
|
+
type: "string",
|
|
1420
|
+
description: "Either 'select:<comma-separated names>' to fetch specific tools, or free-text keywords (e.g. 'github pull request') to rank-search deferred tools."
|
|
1421
|
+
},
|
|
1422
|
+
max_results: {
|
|
1423
|
+
type: "number",
|
|
1424
|
+
description: `Maximum number of tools to return for keyword search. Defaults to ${DEFAULT_MAX_RESULTS}. Ignored for 'select:' queries.`
|
|
1425
|
+
}
|
|
1426
|
+
},
|
|
1427
|
+
required: ["query"]
|
|
1428
|
+
}
|
|
1429
|
+
},
|
|
1430
|
+
toolFn: async (params) => {
|
|
1431
|
+
const parsedParams = ToolSearchParamsSchema.safeParse(params ?? {});
|
|
1432
|
+
if (!parsedParams.success) {
|
|
1433
|
+
const issue = parsedParams.error.issues[0];
|
|
1434
|
+
return `tool_search: invalid parameters — ${issue.path.join(".") || "params"}: ${issue.message}`;
|
|
1435
|
+
}
|
|
1436
|
+
const { query, max_results } = parsedParams.data;
|
|
1437
|
+
const parsed = parseQuery(query);
|
|
1438
|
+
let matched;
|
|
1439
|
+
let unmatched = [];
|
|
1440
|
+
if (parsed.mode === "select") {
|
|
1441
|
+
matched = deferredToolRegistry.getByNames(parsed.names);
|
|
1442
|
+
const foundNames = new Set(matched.map((t) => t.toolSchema.name));
|
|
1443
|
+
unmatched = parsed.names.filter((n) => !foundNames.has(n));
|
|
1444
|
+
} else {
|
|
1445
|
+
const max = max_results ?? DEFAULT_MAX_RESULTS;
|
|
1446
|
+
matched = deferredToolRegistry.searchByKeywords(parsed.text, max);
|
|
1447
|
+
}
|
|
1448
|
+
if (matched.length === 0) return parsed.mode === "select" ? `tool_search: no deferred tools matched ${parsed.names.join(", ")}. Use a free-text query to search.` : `tool_search: no deferred tools matched query "${parsed.text}".`;
|
|
1449
|
+
const liveTools = toolListAccessor();
|
|
1450
|
+
const liveNames = new Set(liveTools.map((t) => t.toolSchema.name));
|
|
1451
|
+
let added = 0;
|
|
1452
|
+
for (const tool of matched) if (!liveNames.has(tool.toolSchema.name)) {
|
|
1453
|
+
liveTools.push(tool);
|
|
1454
|
+
added++;
|
|
1455
|
+
}
|
|
1456
|
+
logger.debug(`[tool_search] query="${query}" matched=${matched.length} added=${added} alreadyLoaded=${matched.length - added}`);
|
|
1457
|
+
const block = renderToolsBlock(matched);
|
|
1458
|
+
return `${`Loaded ${added} new tool schema(s)${added < matched.length ? ` (${matched.length - added} already loaded)` : ""}. These are now callable in your next message.${unmatched.length > 0 ? `\n\nNot found: ${unmatched.join(", ")}` : ""}`}\n\n${block}`;
|
|
1459
|
+
}
|
|
1460
|
+
};
|
|
1461
|
+
}
|
|
1462
|
+
//#endregion
|
|
1463
|
+
//#region src/llm/MultiLlmBackend.ts
|
|
1464
|
+
/**
|
|
1465
|
+
* Routes completions between B4M server and a local Ollama instance
|
|
1466
|
+
* based on the selected model's backend type.
|
|
1467
|
+
*/
|
|
1468
|
+
var MultiLlmBackend = class {
|
|
1469
|
+
constructor(serverBackend, ollamaBackend, serverModels, ollamaModels, initialModel) {
|
|
1470
|
+
this.serverBackend = serverBackend;
|
|
1471
|
+
this.ollamaBackend = ollamaBackend;
|
|
1472
|
+
this.serverModels = serverModels;
|
|
1473
|
+
this.ollamaModels = ollamaModels;
|
|
1474
|
+
this.currentModel = initialModel;
|
|
1475
|
+
this.ollamaModelIds = new Set(ollamaModels.map((m) => m.id));
|
|
1476
|
+
}
|
|
1477
|
+
get activeBackend() {
|
|
1478
|
+
return this.ollamaModelIds.has(this.currentModel) ? this.ollamaBackend : this.serverBackend;
|
|
1479
|
+
}
|
|
1480
|
+
async complete(model, messages, options, callback) {
|
|
1481
|
+
return (this.ollamaModelIds.has(model) ? this.ollamaBackend : this.serverBackend).complete(model, messages, options, callback);
|
|
1482
|
+
}
|
|
1483
|
+
pushToolMessages(messages, tool, result, thinkingBlocks) {
|
|
1484
|
+
this.activeBackend.pushToolMessages(messages, tool, result, thinkingBlocks);
|
|
1485
|
+
}
|
|
1486
|
+
async getModelInfo() {
|
|
1487
|
+
return [...this.serverModels, ...this.ollamaModels];
|
|
1488
|
+
}
|
|
1489
|
+
};
|
|
1490
|
+
//#endregion
|
|
1491
|
+
//#region src/ws/WebSocketConnectionManager.ts
|
|
1492
|
+
const useWsPolyfill = typeof globalThis.WebSocket === "undefined";
|
|
1493
|
+
const WS = useWsPolyfill ? WebSocket : globalThis.WebSocket;
|
|
1494
|
+
/**
|
|
1495
|
+
* Manages a persistent WebSocket connection for CLI <-> server communication.
|
|
1496
|
+
* Handles heartbeat, reconnection, and message routing by requestId.
|
|
1497
|
+
*
|
|
1498
|
+
* Uses Node.js built-in WebSocket (Node 22+) with `ws` package fallback for Node 20.
|
|
1499
|
+
*/
|
|
1500
|
+
var WebSocketConnectionManager = class {
|
|
1501
|
+
/**
|
|
1502
|
+
* @param verifySession - Optional. Called when a connect ATTEMPT fails (the socket closed
|
|
1503
|
+
* without ever opening) - exactly the signal an auth-rejected handshake produces. Omit to
|
|
1504
|
+
* preserve the old always-retry-forever behavior.
|
|
1505
|
+
*/
|
|
1506
|
+
constructor(wsUrl, getToken, verifySession) {
|
|
1507
|
+
this.ws = null;
|
|
1508
|
+
this.heartbeatInterval = null;
|
|
1509
|
+
this.reconnectAttempts = 0;
|
|
1510
|
+
this.maxReconnectDelay = 3e4;
|
|
1511
|
+
this.handlers = /* @__PURE__ */ new Map();
|
|
1512
|
+
this.actionHandlers = /* @__PURE__ */ new Map();
|
|
1513
|
+
this.disconnectHandlers = /* @__PURE__ */ new Set();
|
|
1514
|
+
this.revokedHandlers = /* @__PURE__ */ new Set();
|
|
1515
|
+
this.reconnectTimer = null;
|
|
1516
|
+
this.connected = false;
|
|
1517
|
+
this.connecting = false;
|
|
1518
|
+
this.closed = false;
|
|
1519
|
+
this.openedThisAttempt = false;
|
|
1520
|
+
this.verifyingSession = false;
|
|
1521
|
+
this.revoked = false;
|
|
1522
|
+
this.wsUrl = wsUrl;
|
|
1523
|
+
this.getToken = getToken;
|
|
1524
|
+
this.verifySession = verifySession;
|
|
1525
|
+
}
|
|
1526
|
+
/**
|
|
1527
|
+
* Connect to the WebSocket server.
|
|
1528
|
+
* Resolves when connection is established, rejects on failure.
|
|
1529
|
+
*/
|
|
1530
|
+
async connect() {
|
|
1531
|
+
if (this.connected || this.connecting) return;
|
|
1532
|
+
this.connecting = true;
|
|
1533
|
+
this.openedThisAttempt = false;
|
|
1534
|
+
const token = await this.getToken();
|
|
1535
|
+
if (!token) {
|
|
1536
|
+
this.connecting = false;
|
|
1537
|
+
throw new Error("No access token available for WebSocket connection");
|
|
1538
|
+
}
|
|
1539
|
+
return new Promise((resolve, reject) => {
|
|
1540
|
+
logger.debug(`[WS] Connecting to ${this.wsUrl}...`);
|
|
1541
|
+
if (useWsPolyfill) this.ws = new WebSocket(this.wsUrl, { headers: { "Sec-WebSocket-Protocol": `access_token.${token}` } });
|
|
1542
|
+
else this.ws = new WS(this.wsUrl, [`access_token.${token}`]);
|
|
1543
|
+
this.ws.onopen = () => {
|
|
1544
|
+
logger.debug("[WS] Connected");
|
|
1545
|
+
this.connected = true;
|
|
1546
|
+
this.connecting = false;
|
|
1547
|
+
this.openedThisAttempt = true;
|
|
1548
|
+
this.reconnectAttempts = 0;
|
|
1549
|
+
this.startHeartbeat();
|
|
1550
|
+
resolve();
|
|
1551
|
+
};
|
|
1552
|
+
this.ws.onmessage = (event) => {
|
|
1553
|
+
try {
|
|
1554
|
+
const data = typeof event.data === "string" ? event.data : event.data.toString();
|
|
1555
|
+
const message = JSON.parse(data);
|
|
1556
|
+
const requestId = message.requestId;
|
|
1557
|
+
if (requestId && this.handlers.has(requestId)) this.handlers.get(requestId)(message);
|
|
1558
|
+
else {
|
|
1559
|
+
const action = message.action;
|
|
1560
|
+
if (action && this.actionHandlers.has(action)) this.actionHandlers.get(action)(message);
|
|
1561
|
+
else logger.debug(`[WS] Unhandled message: ${action || "unknown"}`);
|
|
1562
|
+
}
|
|
1563
|
+
} catch (err) {
|
|
1564
|
+
logger.debug(`[WS] Failed to parse message: ${err}`);
|
|
1565
|
+
}
|
|
1566
|
+
};
|
|
1567
|
+
this.ws.onclose = () => {
|
|
1568
|
+
logger.debug("[WS] Connection closed");
|
|
1569
|
+
const openedThisAttempt = this.openedThisAttempt;
|
|
1570
|
+
this.cleanup();
|
|
1571
|
+
this.notifyDisconnect();
|
|
1572
|
+
if (this.closed || this.revoked) return;
|
|
1573
|
+
if (openedThisAttempt || !this.verifySession) {
|
|
1574
|
+
this.scheduleReconnect();
|
|
1575
|
+
return;
|
|
1576
|
+
}
|
|
1577
|
+
this.verifyThenReconnect();
|
|
1578
|
+
};
|
|
1579
|
+
this.ws.onerror = (err) => {
|
|
1580
|
+
const detail = err.error?.message || String(err);
|
|
1581
|
+
logger.debug(`[WS] Error: ${detail}`);
|
|
1582
|
+
if (this.connecting) {
|
|
1583
|
+
this.connecting = false;
|
|
1584
|
+
this.connected = false;
|
|
1585
|
+
reject(/* @__PURE__ */ new Error(`WebSocket connection failed: ${detail}`));
|
|
1586
|
+
}
|
|
1587
|
+
};
|
|
1588
|
+
});
|
|
1589
|
+
}
|
|
1590
|
+
/** Whether the connection is currently established */
|
|
1591
|
+
get isConnected() {
|
|
1592
|
+
return this.connected;
|
|
1593
|
+
}
|
|
1594
|
+
/**
|
|
1595
|
+
* Send a JSON message over the WebSocket connection.
|
|
1596
|
+
*/
|
|
1597
|
+
send(data) {
|
|
1598
|
+
if (!this.ws || this.ws.readyState !== WS.OPEN) throw new Error("WebSocket is not connected");
|
|
1599
|
+
const payload = JSON.stringify(data);
|
|
1600
|
+
const sizeKB = (payload.length / 1024).toFixed(1);
|
|
1601
|
+
logger.debug(`[WS] Sending ${sizeKB} KB (action: ${data.action})`);
|
|
1602
|
+
if (payload.length > 32e3) logger.warn(`[WS] Payload ${sizeKB} KB exceeds API Gateway 32 KB frame limit — connection will be closed`);
|
|
1603
|
+
this.ws.send(payload);
|
|
1604
|
+
}
|
|
1605
|
+
/**
|
|
1606
|
+
* Register a handler for messages matching a specific requestId.
|
|
1607
|
+
*/
|
|
1608
|
+
onRequest(requestId, handler) {
|
|
1609
|
+
this.handlers.set(requestId, handler);
|
|
1610
|
+
}
|
|
1611
|
+
/**
|
|
1612
|
+
* Remove a handler for a specific requestId.
|
|
1613
|
+
*/
|
|
1614
|
+
offRequest(requestId) {
|
|
1615
|
+
this.handlers.delete(requestId);
|
|
1616
|
+
}
|
|
1617
|
+
/**
|
|
1618
|
+
* Register a handler for messages matching a specific action type.
|
|
1619
|
+
* Used for server-pushed commands like keep_command.
|
|
1620
|
+
*/
|
|
1621
|
+
onAction(action, handler) {
|
|
1622
|
+
this.actionHandlers.set(action, handler);
|
|
1623
|
+
}
|
|
1624
|
+
/**
|
|
1625
|
+
* Remove a handler for a specific action type.
|
|
1626
|
+
*/
|
|
1627
|
+
offAction(action) {
|
|
1628
|
+
this.actionHandlers.delete(action);
|
|
1629
|
+
}
|
|
1630
|
+
/**
|
|
1631
|
+
* Register a handler that fires when the connection drops.
|
|
1632
|
+
*/
|
|
1633
|
+
onDisconnect(handler) {
|
|
1634
|
+
this.disconnectHandlers.add(handler);
|
|
1635
|
+
}
|
|
1636
|
+
/**
|
|
1637
|
+
* Remove a disconnect handler.
|
|
1638
|
+
*/
|
|
1639
|
+
offDisconnect(handler) {
|
|
1640
|
+
this.disconnectHandlers.delete(handler);
|
|
1641
|
+
}
|
|
1642
|
+
/**
|
|
1643
|
+
* Register a handler that fires once the session is confirmed revoked (verifySession
|
|
1644
|
+
* returned false) and the reconnect loop has permanently stopped.
|
|
1645
|
+
*/
|
|
1646
|
+
onRevoked(handler) {
|
|
1647
|
+
this.revokedHandlers.add(handler);
|
|
1648
|
+
}
|
|
1649
|
+
/**
|
|
1650
|
+
* Remove a revoked handler.
|
|
1651
|
+
*/
|
|
1652
|
+
offRevoked(handler) {
|
|
1653
|
+
this.revokedHandlers.delete(handler);
|
|
1654
|
+
}
|
|
1655
|
+
/** Whether the session has been confirmed revoked (reconnecting has permanently stopped). */
|
|
1656
|
+
get isRevoked() {
|
|
1657
|
+
return this.revoked;
|
|
1658
|
+
}
|
|
1659
|
+
/**
|
|
1660
|
+
* Close the connection and stop all heartbeat/reconnect logic.
|
|
1661
|
+
*/
|
|
1662
|
+
disconnect() {
|
|
1663
|
+
this.closed = true;
|
|
1664
|
+
this.cleanup();
|
|
1665
|
+
if (this.ws) {
|
|
1666
|
+
this.ws.close();
|
|
1667
|
+
this.ws = null;
|
|
1668
|
+
}
|
|
1669
|
+
this.handlers.clear();
|
|
1670
|
+
this.actionHandlers.clear();
|
|
1671
|
+
this.disconnectHandlers.clear();
|
|
1672
|
+
this.revokedHandlers.clear();
|
|
1673
|
+
}
|
|
1674
|
+
startHeartbeat() {
|
|
1675
|
+
this.stopHeartbeat();
|
|
1676
|
+
this.heartbeatInterval = setInterval(() => {
|
|
1677
|
+
if (this.ws && this.ws.readyState === WS.OPEN) {
|
|
1678
|
+
this.ws.send(JSON.stringify({ action: "heartbeat" }));
|
|
1679
|
+
logger.debug("[WS] Heartbeat sent");
|
|
1680
|
+
}
|
|
1681
|
+
}, 3e5);
|
|
1682
|
+
}
|
|
1683
|
+
stopHeartbeat() {
|
|
1684
|
+
if (this.heartbeatInterval) {
|
|
1685
|
+
clearInterval(this.heartbeatInterval);
|
|
1686
|
+
this.heartbeatInterval = null;
|
|
1687
|
+
}
|
|
1688
|
+
}
|
|
1689
|
+
cleanup() {
|
|
1690
|
+
this.connected = false;
|
|
1691
|
+
this.connecting = false;
|
|
1692
|
+
this.stopHeartbeat();
|
|
1693
|
+
if (this.reconnectTimer) {
|
|
1694
|
+
clearTimeout(this.reconnectTimer);
|
|
1695
|
+
this.reconnectTimer = null;
|
|
1696
|
+
}
|
|
1697
|
+
}
|
|
1698
|
+
notifyDisconnect() {
|
|
1699
|
+
for (const handler of this.disconnectHandlers) try {
|
|
1700
|
+
handler();
|
|
1701
|
+
} catch {}
|
|
1702
|
+
}
|
|
1703
|
+
notifyRevoked() {
|
|
1704
|
+
for (const handler of this.revokedHandlers) try {
|
|
1705
|
+
handler();
|
|
1706
|
+
} catch {}
|
|
1707
|
+
}
|
|
1708
|
+
/**
|
|
1709
|
+
* Called when a connect attempt fails to open at all - the signal a 401 handshake refusal
|
|
1710
|
+
* produces. Verifies the session via the injected `verifySession` (single-flighted) before
|
|
1711
|
+
* deciding whether to keep retrying. A verification that itself errors (network blip, 5xx)
|
|
1712
|
+
* is treated as transient - only an explicit `false` result stops the loop.
|
|
1713
|
+
*/
|
|
1714
|
+
async verifyThenReconnect() {
|
|
1715
|
+
if (this.verifyingSession) return;
|
|
1716
|
+
this.verifyingSession = true;
|
|
1717
|
+
try {
|
|
1718
|
+
if (!await this.verifySession()) {
|
|
1719
|
+
logger.debug("[WS] Session verification failed - session revoked, stopping reconnect");
|
|
1720
|
+
this.revoked = true;
|
|
1721
|
+
this.notifyRevoked();
|
|
1722
|
+
return;
|
|
1723
|
+
}
|
|
1724
|
+
} catch (err) {
|
|
1725
|
+
logger.debug(`[WS] Session verification errored - treating as transient: ${err}`);
|
|
1726
|
+
} finally {
|
|
1727
|
+
this.verifyingSession = false;
|
|
1728
|
+
}
|
|
1729
|
+
this.scheduleReconnect();
|
|
1730
|
+
}
|
|
1731
|
+
scheduleReconnect() {
|
|
1732
|
+
if (this.closed || this.revoked) return;
|
|
1733
|
+
this.reconnectAttempts++;
|
|
1734
|
+
const delay = Math.min(1e3 * Math.pow(2, this.reconnectAttempts - 1), this.maxReconnectDelay);
|
|
1735
|
+
logger.debug(`[WS] Reconnecting in ${delay}ms (attempt ${this.reconnectAttempts})`);
|
|
1736
|
+
this.reconnectTimer = setTimeout(async () => {
|
|
1737
|
+
this.reconnectTimer = null;
|
|
1738
|
+
if (this.closed) return;
|
|
1739
|
+
try {
|
|
1740
|
+
await this.connect();
|
|
1741
|
+
} catch {
|
|
1742
|
+
logger.debug("[WS] Reconnection failed");
|
|
1743
|
+
}
|
|
1744
|
+
}, delay);
|
|
1745
|
+
}
|
|
1746
|
+
};
|
|
1747
|
+
//#endregion
|
|
1748
|
+
//#region src/bootstrap/buildLlmBackend.ts
|
|
1749
|
+
/** Production wiring: real transport classes + the ToolRouter singleton. */
|
|
1750
|
+
const defaultLlmBackendDeps = {
|
|
1751
|
+
connectWebSocket: async (wsUrl, tokenGetter, verifySession) => {
|
|
1752
|
+
const ws = new WebSocketConnectionManager(wsUrl, tokenGetter, verifySession);
|
|
1753
|
+
ws.onRevoked(() => {
|
|
1754
|
+
logger.warn("Session revoked - run `b4m login` again. WebSocket reconnect stopped.");
|
|
1755
|
+
});
|
|
1756
|
+
try {
|
|
1757
|
+
await ws.connect();
|
|
1758
|
+
} catch (err) {
|
|
1759
|
+
ws.disconnect();
|
|
1760
|
+
throw err;
|
|
1761
|
+
}
|
|
1762
|
+
return ws;
|
|
1763
|
+
},
|
|
1764
|
+
clearWebSocketToolExecutor: () => setWebSocketToolExecutor(null),
|
|
1765
|
+
createServerBackend: (opts) => new ServerLlmBackend(opts),
|
|
1766
|
+
createOllamaBackend: (host) => new OllamaBackend(host, {
|
|
1767
|
+
debug: (...args) => logger.debug(args.map(String).join(" ")),
|
|
1768
|
+
info: (...args) => logger.info(args.map(String).join(" ")),
|
|
1769
|
+
warn: (...args) => logger.warn(args.map(String).join(" ")),
|
|
1770
|
+
error: (...args) => logger.error(args.map(String).join(" "))
|
|
1771
|
+
}),
|
|
1772
|
+
createMultiBackend: (server, ollama, serverModels, ollamaModels, defaultModel) => new MultiLlmBackend(server, ollama, serverModels, ollamaModels, defaultModel)
|
|
1773
|
+
};
|
|
1774
|
+
/**
|
|
1775
|
+
* True when some enabled feature module consumes realtime server events. Only
|
|
1776
|
+
* Tavern does today (TavernModule.registerWsHandlers -> TavernActivityStream);
|
|
1777
|
+
* keep this in sync with the module registration in index.tsx. Everything else
|
|
1778
|
+
* runs socket-free, so the common path never opens a WebSocket.
|
|
1779
|
+
*/
|
|
1780
|
+
function needsFeatureEventSocket(config) {
|
|
1781
|
+
return config.features?.tavern === true;
|
|
1782
|
+
}
|
|
1783
|
+
/**
|
|
1784
|
+
* Connect the events-only socket that feature modules register handlers on.
|
|
1785
|
+
* Returns null when it isn't needed, isn't advertised, or won't connect - a
|
|
1786
|
+
* feature's live updates degrading is never a reason to fail startup, since
|
|
1787
|
+
* completions no longer depend on this socket at all.
|
|
1788
|
+
*/
|
|
1789
|
+
async function connectFeatureEventSocket(config, websocketUrl, deps, auth) {
|
|
1790
|
+
if (!needsFeatureEventSocket(config)) return null;
|
|
1791
|
+
if (!websocketUrl) {
|
|
1792
|
+
logger.debug("[WS] No websocketUrl in server config - feature live updates disabled");
|
|
1793
|
+
return null;
|
|
1794
|
+
}
|
|
1795
|
+
try {
|
|
1796
|
+
return await deps.connectWebSocket(websocketUrl, auth.tokenGetter, auth.verifySession);
|
|
1797
|
+
} catch (err) {
|
|
1798
|
+
logger.warn(`Realtime socket unavailable - live feature updates are disabled: ${err instanceof Error ? err.message : String(err)}`);
|
|
1799
|
+
return null;
|
|
1800
|
+
}
|
|
1801
|
+
}
|
|
1802
|
+
/**
|
|
1803
|
+
* Resolve the model to use from the available list: the requested default if
|
|
1804
|
+
* present, otherwise the first available model. Pure - exported for testing.
|
|
1805
|
+
*/
|
|
1806
|
+
function resolveModelInfo(models, defaultModel) {
|
|
1807
|
+
return models.find((m) => m.id === defaultModel) || models[0];
|
|
1808
|
+
}
|
|
1809
|
+
/**
|
|
1810
|
+
* Build the LLM backend: HTTP+SSE transport (ServerLlmBackend), optional Ollama
|
|
1811
|
+
* multiplexing. Resolves the default model and pins it on the backend.
|
|
1812
|
+
*
|
|
1813
|
+
* The WebSocket COMPLETION transport was removed - completions always use SSE,
|
|
1814
|
+
* because relays that emit the generic `streamed_chat_completion` action drop
|
|
1815
|
+
* every CLI chunk. The socket itself is still connected, but only when a
|
|
1816
|
+
* WS-consuming feature module is enabled, and only to carry that module's events
|
|
1817
|
+
* (see `wsManager`); Keep relay and WS server-side tool execution stay off.
|
|
1818
|
+
*
|
|
1819
|
+
* Pure bootstrap seam: no React hooks, no Zustand state.
|
|
1820
|
+
*/
|
|
1821
|
+
async function buildLlmBackend(input, deps = defaultLlmBackendDeps) {
|
|
1822
|
+
const { config, apiClient, startupLog, tokenGetter } = input;
|
|
1823
|
+
const sse = await createSseBackend({
|
|
1824
|
+
apiClient,
|
|
1825
|
+
model: config.defaultModel
|
|
1826
|
+
}, {
|
|
1827
|
+
createServerBackend: deps.createServerBackend,
|
|
1828
|
+
clearWebSocketToolExecutor: deps.clearWebSocketToolExecutor
|
|
1829
|
+
});
|
|
1830
|
+
let llm = sse.llm;
|
|
1831
|
+
const wsManager = await connectFeatureEventSocket(config, sse.serverConfig.websocketUrl, deps, {
|
|
1832
|
+
tokenGetter,
|
|
1833
|
+
verifySession: () => apiClient.checkSessionValid()
|
|
1834
|
+
});
|
|
1835
|
+
const ollamaHost = input.ollamaHost ?? process.env.B4M_OLLAMA_HOST;
|
|
1836
|
+
let models;
|
|
1837
|
+
if (ollamaHost) {
|
|
1838
|
+
const ollamaBackend = deps.createOllamaBackend(ollamaHost);
|
|
1839
|
+
const [serverModels, ollamaModels] = await Promise.all([llm.getModelInfo(), ollamaBackend.getModelInfo()]);
|
|
1840
|
+
if (serverModels.length === 0 && ollamaModels.length === 0) throw new Error(`No models available from server or Ollama at ${ollamaHost}.\nPull a model: ollama pull qwen3.5`);
|
|
1841
|
+
if (ollamaModels.length === 0) startupLog.push(`⚠️ No models found in Ollama at ${ollamaHost}. Pull one with: ollama pull qwen3.5`);
|
|
1842
|
+
const serverBackend = llm;
|
|
1843
|
+
llm = deps.createMultiBackend(serverBackend, ollamaBackend, serverModels, ollamaModels, config.defaultModel);
|
|
1844
|
+
models = await llm.getModelInfo();
|
|
1845
|
+
startupLog.push(`🦙 Self-hosted Ollama: ${ollamaModels.length} model(s) added to picker`);
|
|
1846
|
+
} else {
|
|
1847
|
+
models = await llm.getModelInfo();
|
|
1848
|
+
if (models.length === 0) throw new Error("No models available from server.");
|
|
1849
|
+
}
|
|
1850
|
+
logger.debug(`📋 Available models: ${models.map((m) => m.id).join(", ")}`);
|
|
1851
|
+
const modelInfo = resolveModelInfo(models, config.defaultModel);
|
|
1852
|
+
if (modelInfo.id !== config.defaultModel) {
|
|
1853
|
+
logger.warn(`⚠️ Requested model '${config.defaultModel}' not available`);
|
|
1854
|
+
logger.warn(`🤖 Using fallback model: ${modelInfo.id}`);
|
|
1855
|
+
}
|
|
1856
|
+
llm.currentModel = modelInfo.id;
|
|
1857
|
+
return {
|
|
1858
|
+
llm,
|
|
1859
|
+
wsManager,
|
|
1860
|
+
models,
|
|
1861
|
+
modelInfo
|
|
1862
|
+
};
|
|
1863
|
+
}
|
|
1864
|
+
//#endregion
|
|
1865
|
+
//#region src/bootstrap/buildSupportingStores.ts
|
|
1866
|
+
/**
|
|
1867
|
+
* Build the supporting stores and orchestration the agent needs: CLI tools
|
|
1868
|
+
* (permission-wrapped + server-routed), MCP manager, agent store, context
|
|
1869
|
+
* files, the deferred-tool registry partition, the subagent orchestrator, and
|
|
1870
|
+
* the background-agent manager.
|
|
1871
|
+
*
|
|
1872
|
+
* Pure bootstrap seam: no React hooks, no Zustand state. React-owned values
|
|
1873
|
+
* (permission/user-question prompt functions, the agent context, and the
|
|
1874
|
+
* background-agent status callbacks) are passed in. Tool *assembly* that weaves
|
|
1875
|
+
* the React workflow-store refs (decision/blocker/review-gate tools) stays in
|
|
1876
|
+
* the shell; this module returns only the agent-construction materials.
|
|
1877
|
+
*/
|
|
1878
|
+
async function buildSupportingStores(input) {
|
|
1879
|
+
const { config, llm, modelId, permissionManager, apiClient, configStore, customCommandStore, checkpointStore, sandboxOrchestrator, additionalDirectories, agentContext, promptFn, userQuestionFn, startupLog, silentLogger, onBackgroundStatusChange, onGroupCompletion, onSubagentUsage } = input;
|
|
1880
|
+
const { tools: b4mTools } = await generateCliTools(config.userId, llm, modelId, permissionManager, promptFn, agentContext, configStore, apiClient, void 0, userQuestionFn, checkpointStore, sandboxOrchestrator, additionalDirectories);
|
|
1881
|
+
const mcpManager = new McpManager(config);
|
|
1882
|
+
const builtinAgentsDir = new URL("../agents/defaults/", import.meta.url).pathname;
|
|
1883
|
+
const agentStore = buildProjectAgentStore(builtinAgentsDir, configStore);
|
|
1884
|
+
const [, , contextResult] = await Promise.all([
|
|
1885
|
+
mcpManager.initialize(),
|
|
1886
|
+
agentStore.loadAgents(),
|
|
1887
|
+
loadProjectContext(configStore)
|
|
1888
|
+
]);
|
|
1889
|
+
const mcpTools = wrapTools(mcpManager.getTools(), {
|
|
1890
|
+
permissionManager,
|
|
1891
|
+
showPermissionPrompt: promptFn,
|
|
1892
|
+
agentContext,
|
|
1893
|
+
configStore,
|
|
1894
|
+
apiClient,
|
|
1895
|
+
sandboxOrchestrator,
|
|
1896
|
+
allowedDirectories: additionalDirectories
|
|
1897
|
+
});
|
|
1898
|
+
const deferredB4mToolNames = /* @__PURE__ */ new Set([
|
|
1899
|
+
"math_evaluate",
|
|
1900
|
+
"dice_roll",
|
|
1901
|
+
"current_datetime",
|
|
1902
|
+
"recent_changes",
|
|
1903
|
+
"prompt_enhancement"
|
|
1904
|
+
]);
|
|
1905
|
+
const deferredB4mTools = b4mTools.filter((t) => deferredB4mToolNames.has(t.toolSchema.name));
|
|
1906
|
+
const loadedB4mTools = b4mTools.filter((t) => !deferredB4mToolNames.has(t.toolSchema.name));
|
|
1907
|
+
deferredToolRegistry.register([...mcpTools, ...deferredB4mTools]);
|
|
1908
|
+
if (mcpTools.length > 0) {
|
|
1909
|
+
const serverSummaries = mcpManager.getToolCount().map((s) => `${s.serverName} (${s.count})`).join(", ");
|
|
1910
|
+
startupLog.push(`🛠️ Loaded ${loadedB4mTools.length} B4M + ${mcpTools.length} MCP tool(s, ${deferredB4mTools.length + mcpTools.length} deferred): ${serverSummaries}`);
|
|
1911
|
+
} else {
|
|
1912
|
+
const suffix = deferredB4mTools.length > 0 ? ` (${deferredB4mTools.length} deferred)` : "";
|
|
1913
|
+
startupLog.push(`🛠️ Loaded ${loadedB4mTools.length} B4M tool(s)${suffix}, no MCP tools`);
|
|
1914
|
+
}
|
|
1915
|
+
const agentSummary = agentStore.getSummary();
|
|
1916
|
+
startupLog.push(`🤖 Loaded ${agentSummary.total} agent(s): ${agentSummary.builtin} built-in, ${agentSummary.global} global, ${agentSummary.project} project`);
|
|
1917
|
+
const historyStore = new AgentHistoryStore(config.preferences.subagentHistoryTtlMs ?? 36e5);
|
|
1918
|
+
const orchestrator = new SubagentOrchestrator({
|
|
1919
|
+
userId: config.userId,
|
|
1920
|
+
llm,
|
|
1921
|
+
logger: silentLogger,
|
|
1922
|
+
permissionManager,
|
|
1923
|
+
showPermissionPrompt: promptFn,
|
|
1924
|
+
configStore,
|
|
1925
|
+
apiClient,
|
|
1926
|
+
agentStore,
|
|
1927
|
+
customCommandStore,
|
|
1928
|
+
enableParallelToolExecution: config.preferences.enableParallelToolExecution === true,
|
|
1929
|
+
showUserQuestion: userQuestionFn,
|
|
1930
|
+
checkpointStore,
|
|
1931
|
+
onSubagentUsage,
|
|
1932
|
+
historyStore,
|
|
1933
|
+
sandboxOrchestrator,
|
|
1934
|
+
additionalDirectories
|
|
1935
|
+
});
|
|
1936
|
+
const backgroundManager = new BackgroundAgentManager(orchestrator);
|
|
1937
|
+
backgroundManager.setOnStatusChange(onBackgroundStatusChange);
|
|
1938
|
+
backgroundManager.setOnGroupCompletion(onGroupCompletion);
|
|
1939
|
+
return {
|
|
1940
|
+
mcpManager,
|
|
1941
|
+
agentStore,
|
|
1942
|
+
contextResult,
|
|
1943
|
+
mcpTools,
|
|
1944
|
+
loadedB4mTools,
|
|
1945
|
+
deferredB4mTools,
|
|
1946
|
+
orchestrator,
|
|
1947
|
+
backgroundManager,
|
|
1948
|
+
historyStore
|
|
1949
|
+
};
|
|
1950
|
+
}
|
|
1951
|
+
//#endregion
|
|
1952
|
+
//#region src/bootstrap/buildAgent.ts
|
|
1953
|
+
/**
|
|
1954
|
+
* Construct the main ReAct agent with the system prompt selected by config
|
|
1955
|
+
* variant, wire the tool_search closure to the agent's live tools array, and
|
|
1956
|
+
* record the agent in the shared observation context.
|
|
1957
|
+
*
|
|
1958
|
+
* Pure bootstrap seam: no React hooks, no Zustand state. The interaction-mode
|
|
1959
|
+
* subscription (`useCliStore.subscribe`) stays in the shell and uses the
|
|
1960
|
+
* returned `buildPromptForMode`. Ordering is load-bearing: agent built ->
|
|
1961
|
+
* agentToolsRef wired -> agentContext.currentAgent set, all here, before the
|
|
1962
|
+
* shell registers the subscription (which guards on currentAgent === agent).
|
|
1963
|
+
*/
|
|
1964
|
+
function buildAgent(input) {
|
|
1965
|
+
const { config, modelId, notifyingLlm, allTools, agentContext, agentToolsRef, silentLogger, sessionId, initialInteractionMode, contextContent, agentStore, customCommandStore, enableSkillTool, additionalDirectories, featureModulePrompts } = input;
|
|
1966
|
+
const promptVariant = config.preferences.promptVariant ?? "current";
|
|
1967
|
+
const buildPromptForMode = (mode) => buildSystemPrompt(promptVariant, {
|
|
1968
|
+
contextContent,
|
|
1969
|
+
agentStore,
|
|
1970
|
+
customCommands: customCommandStore.getModelReachableCommands(),
|
|
1971
|
+
enableSkillTool,
|
|
1972
|
+
enableDynamicAgentCreation: config.preferences.enableDynamicAgentCreation === true,
|
|
1973
|
+
additionalDirectories,
|
|
1974
|
+
featureModulePrompts: featureModulePrompts || void 0,
|
|
1975
|
+
planModeFilePath: mode === "plan" ? getPlanModeFilePath(sessionId) : void 0,
|
|
1976
|
+
appendSystemPrompt: process.env.B4M_APPEND_SYSTEM_PROMPT,
|
|
1977
|
+
deferredToolNames: deferredToolRegistry.getDirectoryNames()
|
|
1978
|
+
});
|
|
1979
|
+
const cliSystemPrompt = buildPromptForMode(initialInteractionMode);
|
|
1980
|
+
const maxIterations = config.preferences.maxIterations === null ? 999999 : config.preferences.maxIterations;
|
|
1981
|
+
const agent = new ReActAgent({
|
|
1982
|
+
userId: config.userId,
|
|
1983
|
+
logger: silentLogger,
|
|
1984
|
+
llm: notifyingLlm,
|
|
1985
|
+
model: modelId,
|
|
1986
|
+
tools: allTools,
|
|
1987
|
+
maxIterations,
|
|
1988
|
+
maxTokens: config.preferences.maxTokens,
|
|
1989
|
+
temperature: config.preferences.temperature,
|
|
1990
|
+
systemPrompt: cliSystemPrompt,
|
|
1991
|
+
unknownToolResolver: async (toolName) => deferredToolRegistry.get(toolName) ?? null
|
|
1992
|
+
});
|
|
1993
|
+
agentToolsRef.current = agent.getTools();
|
|
1994
|
+
agentContext.currentAgent = agent;
|
|
1995
|
+
return {
|
|
1996
|
+
agent,
|
|
1997
|
+
buildPromptForMode
|
|
1998
|
+
};
|
|
1999
|
+
}
|
|
2000
|
+
//#endregion
|
|
2001
|
+
export { deferredToolRegistry as a, createToolSearchTool as i, buildSupportingStores as n, NotifyingLlmBackend as o, buildLlmBackend as r, buildAgent as t };
|