mcp-castor 2026.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +487 -0
- package/bin/castor.js +706 -0
- package/index.js +206 -0
- package/package.json +97 -0
- package/skills/canary-test-staging/SKILL.md +24 -0
- package/skills/evo-mutation-rollback/SKILL.md +29 -0
- package/skills/hypothesis-generation/SKILL.md +26 -0
- package/skills/traceback-condensing/SKILL.md +26 -0
- package/src/castor_runner.js +469 -0
- package/src/config.js +1204 -0
- package/src/env.js +10 -0
- package/src/evo_engine.js +214 -0
- package/src/harness/core/events.js +75 -0
- package/src/harness/core/kernel.js +209 -0
- package/src/harness/evo/evaluator.js +156 -0
- package/src/harness/evo/evo_operator.js +550 -0
- package/src/harness/evo/lineage_dag.js +383 -0
- package/src/harness/evo/trace_repair.js +173 -0
- package/src/harness/evo/watchdog.js +72 -0
- package/src/harness/loop_detector.js +135 -0
- package/src/harness/runner.js +1216 -0
- package/src/harness/services/ast_service.js +1813 -0
- package/src/harness/services/event_logger.js +275 -0
- package/src/harness/services/mcp_bridge.js +408 -0
- package/src/harness/services/provider_vllm.js +728 -0
- package/src/harness/services/sandbox_fs.js +1238 -0
- package/src/harness/services/searxng_lifecycle.js +254 -0
- package/src/harness/services/shell_executor.js +264 -0
- package/src/harness/services/shell_validator.js +506 -0
- package/src/harness/services/web_service.js +828 -0
- package/src/platform.js +344 -0
- package/src/repetition_detector.js +139 -0
- package/src/semaphore.js +373 -0
- package/src/server_lifecycle.js +781 -0
- package/src/skills.js +400 -0
- package/src/state_pruner.js +392 -0
- package/src/task_registry.js +1357 -0
- package/src/telemetry.js +638 -0
- package/src/tools.js +997 -0
- package/src/wsl_bridge.js +629 -0
- package/src/wsl_env.js +171 -0
- package/stream_proxy.js +453 -0
|
@@ -0,0 +1,728 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Direct vLLM HTTP/SSE Streaming Client (Castor vLLM Provider)
|
|
3
|
+
*
|
|
4
|
+
* Provides:
|
|
5
|
+
* - Direct HTTP streaming against vLLM (:18020) or Stream Proxy (:18022)
|
|
6
|
+
* - Proactive keep-alive frame handling
|
|
7
|
+
* - OpenAI-compatible function/tool calling parser
|
|
8
|
+
* - Live token velocity (tokens/sec) and TTFT measurement
|
|
9
|
+
* - Reversible Castor plugin binding
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import {
|
|
13
|
+
VLLM_PORT,
|
|
14
|
+
MAX_TOKENS,
|
|
15
|
+
MAX_LEN_HUGE,
|
|
16
|
+
getReasoningEffort,
|
|
17
|
+
STREAM_IDLE_TIMEOUT_MS,
|
|
18
|
+
STREAM_IDLE_TIMEOUT_MS_DEEP,
|
|
19
|
+
STREAM_IDLE_DEPTH_TOKENS,
|
|
20
|
+
MAX_REASONING_TOKENS,
|
|
21
|
+
MODEL,
|
|
22
|
+
BASE_URL,
|
|
23
|
+
STREAM_PROXY_PORT,
|
|
24
|
+
} from "../../config.js";
|
|
25
|
+
|
|
26
|
+
export class VllmProviderService {
|
|
27
|
+
constructor(options = {}) {
|
|
28
|
+
// Engine identity is sourced from config.js (env > ~/.castor/config.json >
|
|
29
|
+
// built-in defaults) so the provider never hardcodes a model name or
|
|
30
|
+
// endpoint that could drift from the rest of the harness.
|
|
31
|
+
this.baseUrl =
|
|
32
|
+
options.baseUrl ||
|
|
33
|
+
(BASE_URL === `http://localhost:${VLLM_PORT}/v1`
|
|
34
|
+
? `http://127.0.0.1:${STREAM_PROXY_PORT}/v1`
|
|
35
|
+
: BASE_URL);
|
|
36
|
+
this.fallbackUrl = options.fallbackUrl || `http://127.0.0.1:${VLLM_PORT}/v1`;
|
|
37
|
+
const rawModel = options.model || MODEL;
|
|
38
|
+
this.model =
|
|
39
|
+
rawModel && rawModel.toLowerCase() === "qwen3.8-27b"
|
|
40
|
+
? "qwen3.8-27b"
|
|
41
|
+
: rawModel;
|
|
42
|
+
this.defaultTemperature = options.temperature ?? 0.0;
|
|
43
|
+
this.defaultMaxTokens = options.maxTokens ?? MAX_TOKENS;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Fetches available models from the vLLM server.
|
|
48
|
+
*
|
|
49
|
+
* Every failure path (network error, non-2xx, unparseable body, missing
|
|
50
|
+
* `data` array) throws an error carrying the upstream status/detail; only a
|
|
51
|
+
* genuine 2xx with a parseable `data` array yields a model list.
|
|
52
|
+
* @returns {Promise<string[]>} The list of model ids.
|
|
53
|
+
* @throws {Error} When the probe fails for any reason.
|
|
54
|
+
*/
|
|
55
|
+
async listModels() {
|
|
56
|
+
let res;
|
|
57
|
+
try {
|
|
58
|
+
res = await fetch(`${this.baseUrl}/models`, {
|
|
59
|
+
signal: AbortSignal.timeout(5000),
|
|
60
|
+
});
|
|
61
|
+
} catch (err) {
|
|
62
|
+
// Network-level failure (connection refused / timeout / DNS): the
|
|
63
|
+
// engine (or proxy) is unreachable. Surface it verbatim — never a
|
|
64
|
+
// fabricated list.
|
|
65
|
+
throw new Error(
|
|
66
|
+
`listModels: vLLM /models probe failed against ${this.baseUrl}: ${
|
|
67
|
+
err && err.message ? err.message : String(err)
|
|
68
|
+
}`
|
|
69
|
+
);
|
|
70
|
+
}
|
|
71
|
+
if (!res.ok) {
|
|
72
|
+
// Non-2xx: the engine answered but rejected the probe. Carry the
|
|
73
|
+
// upstream status + body so the failure is decidable.
|
|
74
|
+
let detail = "";
|
|
75
|
+
try {
|
|
76
|
+
detail = await res.text();
|
|
77
|
+
} catch {}
|
|
78
|
+
throw new Error(
|
|
79
|
+
`listModels: vLLM /models probe returned HTTP ${res.status} ${res.statusText} from ${this.baseUrl}: ${detail}`
|
|
80
|
+
);
|
|
81
|
+
}
|
|
82
|
+
let data;
|
|
83
|
+
try {
|
|
84
|
+
data = await res.json();
|
|
85
|
+
} catch (err) {
|
|
86
|
+
throw new Error(
|
|
87
|
+
`listModels: vLLM /models returned 200 but an unparseable body from ${this.baseUrl}: ${
|
|
88
|
+
err && err.message ? err.message : String(err)
|
|
89
|
+
}`
|
|
90
|
+
);
|
|
91
|
+
}
|
|
92
|
+
if (!data || !Array.isArray(data.data)) {
|
|
93
|
+
throw new Error(
|
|
94
|
+
`listModels: vLLM /models returned 200 but no 'data' array from ${this.baseUrl} (body: ${JSON.stringify(data)})`
|
|
95
|
+
);
|
|
96
|
+
}
|
|
97
|
+
return data.data.map((m) => m.id);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Streams a chat completion turn from vLLM.
|
|
102
|
+
*
|
|
103
|
+
* @param {object} params
|
|
104
|
+
* @param {Array<object>} params.messages
|
|
105
|
+
* @param {Array<object>} [params.tools] OpenAI-formatted tools
|
|
106
|
+
* @param {number} [params.temperature]
|
|
107
|
+
* @param {number} [params.maxTokens]
|
|
108
|
+
* @param {string} [params.reasoningEffort] Task-local reasoning-effort tier
|
|
109
|
+
* (one of REASONING_EFFORT_TIERS). When provided it overrides the
|
|
110
|
+
* QWEN_REASONING_EFFORT env default for THIS request only — no
|
|
111
|
+
* process.env mutation, so it never leaks across concurrent tasks.
|
|
112
|
+
* @param {AbortSignal} [params.signal]
|
|
113
|
+
* @param {(token: string) => void} [params.onToken]
|
|
114
|
+
* @param {(metric: object) => void} [params.onMetrics]
|
|
115
|
+
* @returns {Promise<{
|
|
116
|
+
* content: string,
|
|
117
|
+
* toolCalls: Array<{ id: string, name: string, arguments: string }>,
|
|
118
|
+
* finishReason: string | null,
|
|
119
|
+
* metrics: { promptTokens: number, completionTokens: number, ttftMs: number, prefillMs: number, generationMs: number, totalMs: number, reasoningTokens: number, hadReasoning: boolean, reasoningCeilingHit: boolean, streamIdleTimeoutMs: number, streamIdleTier: "shallow" | "deep", promptTokensEstimated?: boolean }
|
|
120
|
+
* }>}
|
|
121
|
+
*/
|
|
122
|
+
async streamChat({
|
|
123
|
+
messages,
|
|
124
|
+
tools = [],
|
|
125
|
+
temperature = this.defaultTemperature,
|
|
126
|
+
maxTokens = this.defaultMaxTokens,
|
|
127
|
+
reasoningEffort,
|
|
128
|
+
signal,
|
|
129
|
+
sessionId,
|
|
130
|
+
onToken,
|
|
131
|
+
onMetrics,
|
|
132
|
+
}) {
|
|
133
|
+
const t0 = Date.now();
|
|
134
|
+
|
|
135
|
+
// Dynamic headroom clamping against MAX_LEN_HUGE (245,760)
|
|
136
|
+
const promptChars = JSON.stringify(messages).length + (tools && tools.length > 0 ? JSON.stringify(tools).length : 0);
|
|
137
|
+
const estimatedPromptTokens = Math.ceil(promptChars / 3.5);
|
|
138
|
+
const maxPossibleHeadroom = Math.max(0, MAX_LEN_HUGE - estimatedPromptTokens - 128);
|
|
139
|
+
|
|
140
|
+
if (maxPossibleHeadroom < 1024) {
|
|
141
|
+
throw new Error(
|
|
142
|
+
`ContextExhaustedError: Prompt consumes ~${estimatedPromptTokens} tokens, leaving insufficient headroom (<1024) under model context limit (${MAX_LEN_HUGE}).`
|
|
143
|
+
);
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
const clampedMaxTokens = Math.min(maxTokens, maxPossibleHeadroom);
|
|
147
|
+
|
|
148
|
+
const payload = {
|
|
149
|
+
model: this.model,
|
|
150
|
+
messages,
|
|
151
|
+
stream: true,
|
|
152
|
+
temperature,
|
|
153
|
+
max_tokens: clampedMaxTokens,
|
|
154
|
+
// vLLM 0.28+: request the engine-reported token usage in a terminal
|
|
155
|
+
// SSE chunk (empty choices + a `usage` field). Additive only — engines
|
|
156
|
+
// that ignore it simply never emit the chunk, and we fall back to the
|
|
157
|
+
// chars-based estimate below.
|
|
158
|
+
stream_options: { include_usage: true },
|
|
159
|
+
};
|
|
160
|
+
|
|
161
|
+
if (tools && tools.length > 0) {
|
|
162
|
+
payload.tools = tools;
|
|
163
|
+
payload.tool_choice = "auto";
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
// Reasoning-effort passthrough: a task-local `reasoningEffort` param
|
|
167
|
+
// (threaded from the qwen_coworker dispatch) takes precedence; when it is
|
|
168
|
+
// absent the existing dynamic QWEN_REASONING_EFFORT read is the fallback
|
|
169
|
+
// (unchanged behavior). Forwarded to the vLLM chat template so the engine
|
|
170
|
+
// can trade thinking depth for latency per dispatch. When neither is set,
|
|
171
|
+
// send nothing and let the server default apply.
|
|
172
|
+
const effectiveReasoningEffort = reasoningEffort || getReasoningEffort();
|
|
173
|
+
if (effectiveReasoningEffort) {
|
|
174
|
+
payload.chat_template_kwargs = {
|
|
175
|
+
reasoning_effort: effectiveReasoningEffort,
|
|
176
|
+
};
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
let activeUrl = this.baseUrl;
|
|
180
|
+
let response;
|
|
181
|
+
|
|
182
|
+
// Compose an internal abort controller with the caller's signal so the
|
|
183
|
+
// stream-idle watchdog and the reasoning ceiling can end the request
|
|
184
|
+
// themselves, while an external cancel still propagates.
|
|
185
|
+
const streamController = new AbortController();
|
|
186
|
+
const onExternalAbort = () => streamController.abort(signal?.reason);
|
|
187
|
+
if (signal) {
|
|
188
|
+
if (signal.aborted) streamController.abort(signal.reason);
|
|
189
|
+
else signal.addEventListener("abort", onExternalAbort, { once: true });
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
try {
|
|
193
|
+
try {
|
|
194
|
+
response = await fetch(`${activeUrl}/chat/completions`, {
|
|
195
|
+
method: "POST",
|
|
196
|
+
headers: { "Content-Type": "application/json" },
|
|
197
|
+
body: JSON.stringify(payload),
|
|
198
|
+
signal: streamController.signal,
|
|
199
|
+
});
|
|
200
|
+
} catch (err) {
|
|
201
|
+
// Fallback directly to upstream vLLM port if proxy connection refused.
|
|
202
|
+
// An abort that ORIGINATED from the caller (not a proxy connect
|
|
203
|
+
// failure) must NOT be retried against the fallback.
|
|
204
|
+
if (signal?.aborted) throw err;
|
|
205
|
+
activeUrl = this.fallbackUrl;
|
|
206
|
+
response = await fetch(`${activeUrl}/chat/completions`, {
|
|
207
|
+
method: "POST",
|
|
208
|
+
headers: { "Content-Type": "application/json" },
|
|
209
|
+
body: JSON.stringify(payload),
|
|
210
|
+
signal: streamController.signal,
|
|
211
|
+
});
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
if (!response.ok) {
|
|
215
|
+
const errText = await response.text().catch(() => "");
|
|
216
|
+
throw new Error(`vLLM stream error (${response.status} ${response.statusText}): ${errText}`);
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
return await this._consumeStream({
|
|
220
|
+
response,
|
|
221
|
+
decoder: new TextDecoder("utf-8"),
|
|
222
|
+
t0,
|
|
223
|
+
// Threaded from streamChat (where the chars-based estimate is computed
|
|
224
|
+
// for the context-headroom clamp) so _consumeStream can fall back to it
|
|
225
|
+
// when the engine does not emit a stream_options usage chunk.
|
|
226
|
+
estimatedPromptTokens,
|
|
227
|
+
sessionId,
|
|
228
|
+
onToken,
|
|
229
|
+
onMetrics,
|
|
230
|
+
controller: streamController,
|
|
231
|
+
cleanup: () => {
|
|
232
|
+
if (signal) signal.removeEventListener("abort", onExternalAbort);
|
|
233
|
+
},
|
|
234
|
+
});
|
|
235
|
+
} finally {
|
|
236
|
+
if (signal) signal.removeEventListener("abort", onExternalAbort);
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* Reads the SSE body of an established streaming response.
|
|
242
|
+
*
|
|
243
|
+
* - Stream-idle watchdog: if no meaningful SSE frame arrives within the
|
|
244
|
+
* armed idle window, the request is aborted and the turn fails. The
|
|
245
|
+
* window is STREAM_IDLE_TIMEOUT_MS (shallow) for normal turns, or
|
|
246
|
+
* STREAM_IDLE_TIMEOUT_MS_DEEP when the estimated prompt tokens reach
|
|
247
|
+
* STREAM_IDLE_DEPTH_TOKENS (deep-context turns). The fired tier is named
|
|
248
|
+
* in the thrown error and in metrics.streamIdleTier. Proxy keep-alive
|
|
249
|
+
* comment frames (": keep-alive") do not reset the watchdog.
|
|
250
|
+
* - Reasoning ceiling: when estimated reasoning tokens exceed
|
|
251
|
+
* MAX_REASONING_TOKENS in a single turn, the stream is ended locally with
|
|
252
|
+
* finish_reason "length" and reasoningCeilingHit set.
|
|
253
|
+
* - finish_reason: a stream that ends with no finish_reason and produced
|
|
254
|
+
* neither content nor tool calls is reported as null, not synthesized as
|
|
255
|
+
* "stop".
|
|
256
|
+
*
|
|
257
|
+
* @param {object} params
|
|
258
|
+
* @param {Response} params.response
|
|
259
|
+
* @param {TextDecoder} params.decoder
|
|
260
|
+
* @param {number} params.t0
|
|
261
|
+
* @param {number} params.estimatedPromptTokens
|
|
262
|
+
* @param {string} [params.sessionId]
|
|
263
|
+
* @param {(token: string) => void} [params.onToken]
|
|
264
|
+
* @param {(metric: object) => void} [params.onMetrics]
|
|
265
|
+
* @param {AbortController} params.controller
|
|
266
|
+
* @param {() => void} [params.cleanup]
|
|
267
|
+
* @returns {Promise<{
|
|
268
|
+
* content: string,
|
|
269
|
+
* reasoning: string,
|
|
270
|
+
* toolCalls: Array<{ id: string, name: string, arguments: string }>,
|
|
271
|
+
* finishReason: string | null,
|
|
272
|
+
* metrics: object,
|
|
273
|
+
* reasoningTokens: number,
|
|
274
|
+
* hadReasoning: boolean
|
|
275
|
+
* }>}
|
|
276
|
+
*/
|
|
277
|
+
async _consumeStream({
|
|
278
|
+
response,
|
|
279
|
+
decoder,
|
|
280
|
+
t0,
|
|
281
|
+
estimatedPromptTokens,
|
|
282
|
+
sessionId,
|
|
283
|
+
onToken,
|
|
284
|
+
onMetrics,
|
|
285
|
+
controller,
|
|
286
|
+
cleanup,
|
|
287
|
+
}) {
|
|
288
|
+
const reader = response.body.getReader();
|
|
289
|
+
let buffer = "";
|
|
290
|
+
let fullContent = "";
|
|
291
|
+
let fullReasoning = "";
|
|
292
|
+
let finishReason = null;
|
|
293
|
+
let ttft = null;
|
|
294
|
+
let completionTokens = 0;
|
|
295
|
+
let reasoningTokens = 0;
|
|
296
|
+
let hadReasoning = false;
|
|
297
|
+
let reasoningCeilingHit = false;
|
|
298
|
+
let idleTimedOut = false;
|
|
299
|
+
let idleTimer = null;
|
|
300
|
+
// Engine-reported usage from the terminal stream_options chunk (vLLM 0.28+).
|
|
301
|
+
// Null until the engine emits it; the chars-based estimate is the fallback.
|
|
302
|
+
let engineUsage = null;
|
|
303
|
+
const toolCallsMap = new Map(); // index -> { id, name, arguments }
|
|
304
|
+
|
|
305
|
+
// ------------------------------------------------------------------
|
|
306
|
+
// Inline think-tag normalization (Ollama / llama-server style):
|
|
307
|
+
// engines without a reasoning parser emit thinking segments inside
|
|
308
|
+
// delta.content wrapped in backtick-think tags. A small per-stream
|
|
309
|
+
// state machine splits the stream so the thinking lands in
|
|
310
|
+
// fullReasoning (counted, never surfaced to the user) and only the
|
|
311
|
+
// stripped text reaches fullContent / onToken. The pending buffer
|
|
312
|
+
// holds text whose tag boundary is not yet resolvable (a tag split
|
|
313
|
+
// across chunk boundaries).
|
|
314
|
+
// ------------------------------------------------------------------
|
|
315
|
+
const THINK_OPEN = String.fromCharCode(96) + "think";
|
|
316
|
+
const THINK_CLOSE = String.fromCharCode(96) + "think" + String.fromCharCode(96);
|
|
317
|
+
let thinkPending = "";
|
|
318
|
+
let inThink = false;
|
|
319
|
+
|
|
320
|
+
const flushThinkPending = () => {
|
|
321
|
+
if (!thinkPending) return;
|
|
322
|
+
if (inThink) {
|
|
323
|
+
// Unclosed think segment at stream end: the remainder is reasoning.
|
|
324
|
+
fullReasoning += thinkPending;
|
|
325
|
+
reasoningTokens += Math.max(1, Math.round(thinkPending.length / 4));
|
|
326
|
+
hadReasoning = true;
|
|
327
|
+
} else {
|
|
328
|
+
// A partial opening tag that never completed is plain content.
|
|
329
|
+
fullContent += thinkPending;
|
|
330
|
+
if (onToken) onToken(thinkPending);
|
|
331
|
+
}
|
|
332
|
+
thinkPending = "";
|
|
333
|
+
};
|
|
334
|
+
|
|
335
|
+
const processThinkDelta = (text) => {
|
|
336
|
+
thinkPending += text;
|
|
337
|
+
for (;;) {
|
|
338
|
+
if (inThink) {
|
|
339
|
+
const closeIdx = thinkPending.indexOf(THINK_CLOSE);
|
|
340
|
+
if (closeIdx < 0) return; // hold until the closing tag arrives
|
|
341
|
+
const thinking = thinkPending.slice(0, closeIdx);
|
|
342
|
+
if (thinking.length > 0) {
|
|
343
|
+
fullReasoning += thinking;
|
|
344
|
+
reasoningTokens += Math.max(1, Math.round(thinking.length / 4));
|
|
345
|
+
hadReasoning = true;
|
|
346
|
+
}
|
|
347
|
+
thinkPending = thinkPending.slice(closeIdx + THINK_CLOSE.length);
|
|
348
|
+
inThink = false;
|
|
349
|
+
continue;
|
|
350
|
+
}
|
|
351
|
+
const openIdx = thinkPending.indexOf(THINK_OPEN);
|
|
352
|
+
if (openIdx < 0) {
|
|
353
|
+
// Hold a trailing partial opening tag (e.g., a backtick-th
|
|
354
|
+
// split across chunks) so it can be matched once the rest arrives.
|
|
355
|
+
let hold = 0;
|
|
356
|
+
for (
|
|
357
|
+
let k = Math.min(THINK_OPEN.length, thinkPending.length);
|
|
358
|
+
k > 0;
|
|
359
|
+
k--
|
|
360
|
+
) {
|
|
361
|
+
if (thinkPending.endsWith(THINK_OPEN.slice(0, k))) {
|
|
362
|
+
hold = k;
|
|
363
|
+
break;
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
const emit = thinkPending.slice(0, thinkPending.length - hold);
|
|
367
|
+
if (emit) {
|
|
368
|
+
fullContent += emit;
|
|
369
|
+
if (onToken) onToken(emit);
|
|
370
|
+
}
|
|
371
|
+
thinkPending = thinkPending.slice(thinkPending.length - hold);
|
|
372
|
+
return;
|
|
373
|
+
}
|
|
374
|
+
const closeIdx = thinkPending.indexOf(
|
|
375
|
+
THINK_CLOSE,
|
|
376
|
+
openIdx + THINK_OPEN.length
|
|
377
|
+
);
|
|
378
|
+
const before = thinkPending.slice(0, openIdx);
|
|
379
|
+
if (before) {
|
|
380
|
+
fullContent += before;
|
|
381
|
+
if (onToken) onToken(before);
|
|
382
|
+
}
|
|
383
|
+
if (closeIdx < 0) {
|
|
384
|
+
// Opening tag present but not yet closed: enter think mode and
|
|
385
|
+
// hold the remainder until the closing tag arrives.
|
|
386
|
+
inThink = true;
|
|
387
|
+
thinkPending = thinkPending.slice(openIdx + THINK_OPEN.length);
|
|
388
|
+
return;
|
|
389
|
+
}
|
|
390
|
+
const thinking = thinkPending.slice(
|
|
391
|
+
openIdx + THINK_OPEN.length,
|
|
392
|
+
closeIdx
|
|
393
|
+
);
|
|
394
|
+
if (thinking.length > 0) {
|
|
395
|
+
fullReasoning += thinking;
|
|
396
|
+
reasoningTokens += Math.max(1, Math.round(thinking.length / 4));
|
|
397
|
+
hadReasoning = true;
|
|
398
|
+
}
|
|
399
|
+
thinkPending = thinkPending.slice(closeIdx + THINK_CLOSE.length);
|
|
400
|
+
}
|
|
401
|
+
};
|
|
402
|
+
|
|
403
|
+
// Depth-aware idle tier: a deep-context prompt (estimated prompt tokens
|
|
404
|
+
// >= STREAM_IDLE_DEPTH_TOKENS) gets the longer DEEP window; normal turns
|
|
405
|
+
// keep the SHALLOW window. The tier is chosen from the chars-based
|
|
406
|
+
// estimate already computed in streamChat.
|
|
407
|
+
const isDeep = estimatedPromptTokens >= STREAM_IDLE_DEPTH_TOKENS;
|
|
408
|
+
const idleTimeoutMs = isDeep ? STREAM_IDLE_TIMEOUT_MS_DEEP : STREAM_IDLE_TIMEOUT_MS;
|
|
409
|
+
const idleTier = isDeep ? "deep" : "shallow";
|
|
410
|
+
let deepTierLogged = false;
|
|
411
|
+
|
|
412
|
+
const disarmIdle = () => {
|
|
413
|
+
if (idleTimer) {
|
|
414
|
+
clearTimeout(idleTimer);
|
|
415
|
+
idleTimer = null;
|
|
416
|
+
}
|
|
417
|
+
};
|
|
418
|
+
const armIdle = () => {
|
|
419
|
+
disarmIdle();
|
|
420
|
+
idleTimer = setTimeout(() => {
|
|
421
|
+
idleTimedOut = true;
|
|
422
|
+
controller.abort(new Error(`stream idle > ${idleTimeoutMs}ms (${idleTier} tier)`));
|
|
423
|
+
}, idleTimeoutMs);
|
|
424
|
+
if (typeof idleTimer.unref === "function") idleTimer.unref();
|
|
425
|
+
// Log once when the deep tier arms.
|
|
426
|
+
if (isDeep && !deepTierLogged) {
|
|
427
|
+
deepTierLogged = true;
|
|
428
|
+
console.error(
|
|
429
|
+
`[VllmProvider] Deep-context turn (est. prompt ${estimatedPromptTokens} tokens >= ${STREAM_IDLE_DEPTH_TOKENS}): arming ${idleTimeoutMs}ms (${idleTier} tier) stream-idle watchdog.`
|
|
430
|
+
);
|
|
431
|
+
}
|
|
432
|
+
};
|
|
433
|
+
|
|
434
|
+
let currentSseEvent = null;
|
|
435
|
+
try {
|
|
436
|
+
armIdle();
|
|
437
|
+
while (true) {
|
|
438
|
+
let readResult;
|
|
439
|
+
try {
|
|
440
|
+
readResult = await reader.read();
|
|
441
|
+
} catch (err) {
|
|
442
|
+
if (idleTimedOut) {
|
|
443
|
+
const sid = sessionId || "SESSION_ID";
|
|
444
|
+
const inspectCmd = `node mcp-castor/src/harness/services/event_logger.js ${sid} 5`;
|
|
445
|
+
throw new Error(
|
|
446
|
+
`vLLM stream idle timeout (${idleTier} tier, ${idleTimeoutMs}ms): no meaningful SSE tokens emitted within window.\n` +
|
|
447
|
+
`[Orchestrator Advisory]: There might have been an issue (e.g. extended GPU contention or deliberation).\n` +
|
|
448
|
+
`You can run:\n` +
|
|
449
|
+
` ${inspectCmd}\n` +
|
|
450
|
+
`to inspect the last few traces, and decide whether to roll into a new session (e.g. '${sid}_stage2'), or use the same session and same effort to finish/continue.`
|
|
451
|
+
);
|
|
452
|
+
}
|
|
453
|
+
throw err;
|
|
454
|
+
}
|
|
455
|
+
const { done, value } = readResult;
|
|
456
|
+
if (done) break;
|
|
457
|
+
|
|
458
|
+
buffer += decoder.decode(value, { stream: true });
|
|
459
|
+
const lines = buffer.split("\n");
|
|
460
|
+
buffer = lines.pop(); // keep last incomplete line
|
|
461
|
+
|
|
462
|
+
for (const line of lines) {
|
|
463
|
+
const trimmed = line.trim();
|
|
464
|
+
if (!trimmed || trimmed.startsWith(":")) {
|
|
465
|
+
// SSE comment / keep-alive heartbeat frame — NOT engine activity.
|
|
466
|
+
continue;
|
|
467
|
+
}
|
|
468
|
+
|
|
469
|
+
// Any meaningful SSE frame proves the engine (or an interceptor)
|
|
470
|
+
// is alive and producing; reset the idle watchdog.
|
|
471
|
+
armIdle();
|
|
472
|
+
|
|
473
|
+
if (trimmed.startsWith("event: ")) {
|
|
474
|
+
currentSseEvent = trimmed.slice(7).trim();
|
|
475
|
+
continue;
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
if (trimmed === "data: [DONE]") {
|
|
479
|
+
break;
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
if (trimmed.startsWith("data: ")) {
|
|
483
|
+
const jsonStr = trimmed.slice(6);
|
|
484
|
+
if (currentSseEvent === "error") {
|
|
485
|
+
let errMsg = jsonStr;
|
|
486
|
+
try {
|
|
487
|
+
const parsedErr = JSON.parse(jsonStr);
|
|
488
|
+
errMsg = parsedErr.error?.message || parsedErr.message || jsonStr;
|
|
489
|
+
} catch {}
|
|
490
|
+
throw new Error(`vLLM upstream SSE error: ${errMsg}`);
|
|
491
|
+
}
|
|
492
|
+
currentSseEvent = null;
|
|
493
|
+
|
|
494
|
+
try {
|
|
495
|
+
const chunk = JSON.parse(jsonStr);
|
|
496
|
+
if (chunk.error && !chunk.choices) {
|
|
497
|
+
const errMsg = chunk.error.message || JSON.stringify(chunk.error);
|
|
498
|
+
throw new Error(`vLLM upstream error: ${errMsg}`);
|
|
499
|
+
}
|
|
500
|
+
if (chunk.id === "chatcmpl-stream-err") {
|
|
501
|
+
const errMsg = chunk.choices?.[0]?.delta?.content || "vLLM stream error";
|
|
502
|
+
throw new Error(`vLLM stream error: ${errMsg}`);
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
// stream_options { include_usage: true } (vLLM 0.28+): the engine
|
|
506
|
+
// emits a terminal chunk whose `choices` is EMPTY and whose
|
|
507
|
+
// `usage` field carries the authoritative prompt/completion token
|
|
508
|
+
// counts. Capture it here, before the `choice` guard below, so the
|
|
509
|
+
// empty-choices chunk is consumed gracefully and does not
|
|
510
|
+
// contribute to the finish-reason or content signals.
|
|
511
|
+
if (chunk.usage && typeof chunk.usage === "object") {
|
|
512
|
+
engineUsage = chunk.usage;
|
|
513
|
+
}
|
|
514
|
+
|
|
515
|
+
const choice = chunk.choices?.[0];
|
|
516
|
+
if (!choice) continue;
|
|
517
|
+
|
|
518
|
+
if (choice.finish_reason) {
|
|
519
|
+
finishReason = choice.finish_reason;
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
const delta = choice.delta;
|
|
523
|
+
if (!delta) continue;
|
|
524
|
+
|
|
525
|
+
// Server-side reasoning (thinking) is streamed by vLLM's
|
|
526
|
+
// --reasoning-parser qwen3 as delta.reasoning, while some engines
|
|
527
|
+
// / older parsers use delta.reasoning_content. Normalize both so
|
|
528
|
+
// accounting works regardless of which the engine emits. It is
|
|
529
|
+
// accounted for (so the ledger shows thinking volume and TTFT
|
|
530
|
+
// reflects the first output of any kind) but is not appended to
|
|
531
|
+
// fullContent or forwarded to onToken — the user-visible token
|
|
532
|
+
// stream stays clean, and the engine-side /metrics token counters
|
|
533
|
+
// already cover watchdog velocity.
|
|
534
|
+
const reasoningText =
|
|
535
|
+
typeof delta.reasoning === "string"
|
|
536
|
+
? delta.reasoning
|
|
537
|
+
: typeof delta.reasoning_content === "string"
|
|
538
|
+
? delta.reasoning_content
|
|
539
|
+
: null;
|
|
540
|
+
if (reasoningText && reasoningText.length > 0) {
|
|
541
|
+
hadReasoning = true;
|
|
542
|
+
fullReasoning += reasoningText;
|
|
543
|
+
// ~chars/4 is a reasonable token estimate for thinking text.
|
|
544
|
+
reasoningTokens += Math.max(1, Math.round(reasoningText.length / 4));
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
// Measure Time to First Token: the first delta of ANY kind
|
|
548
|
+
// (content, tool call, or reasoning) means generation started.
|
|
549
|
+
if (
|
|
550
|
+
ttft === null &&
|
|
551
|
+
(delta.content || delta.tool_calls || reasoningText)
|
|
552
|
+
) {
|
|
553
|
+
ttft = Date.now() - t0;
|
|
554
|
+
}
|
|
555
|
+
|
|
556
|
+
// Stream text content. Engines without a reasoning parser
|
|
557
|
+
// (Ollama, llama-server) emit think-tag segments inside
|
|
558
|
+
// delta.content; the state machine above splits them so the
|
|
559
|
+
// thinking lands in fullReasoning and only the stripped text
|
|
560
|
+
// reaches fullContent / onToken.
|
|
561
|
+
if (delta.content) {
|
|
562
|
+
completionTokens++;
|
|
563
|
+
processThinkDelta(delta.content);
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
// Stream tool calls
|
|
567
|
+
if (delta.tool_calls && Array.isArray(delta.tool_calls)) {
|
|
568
|
+
for (const tc of delta.tool_calls) {
|
|
569
|
+
const idx = tc.index ?? 0;
|
|
570
|
+
if (!toolCallsMap.has(idx)) {
|
|
571
|
+
toolCallsMap.set(idx, {
|
|
572
|
+
id: tc.id || `call_${Date.now()}_${idx}`,
|
|
573
|
+
name: tc.function?.name || "",
|
|
574
|
+
arguments: tc.function?.arguments || "",
|
|
575
|
+
});
|
|
576
|
+
} else {
|
|
577
|
+
const existing = toolCallsMap.get(idx);
|
|
578
|
+
if (tc.id && !existing.id) existing.id = tc.id;
|
|
579
|
+
if (tc.function?.name) existing.name += tc.function.name;
|
|
580
|
+
if (tc.function?.arguments) existing.arguments += tc.function.arguments;
|
|
581
|
+
}
|
|
582
|
+
}
|
|
583
|
+
}
|
|
584
|
+
|
|
585
|
+
// Reasoning-ceiling enforcement: cutting the stream here reports
|
|
586
|
+
// as a length cutoff (with hadReasoning already true), which
|
|
587
|
+
// routes the runner into the reasoning-cutoff continuation
|
|
588
|
+
// directive instead of a retry/failure path.
|
|
589
|
+
if (
|
|
590
|
+
!reasoningCeilingHit &&
|
|
591
|
+
reasoningTokens >= MAX_REASONING_TOKENS
|
|
592
|
+
) {
|
|
593
|
+
reasoningCeilingHit = true;
|
|
594
|
+
finishReason = "length";
|
|
595
|
+
console.error(
|
|
596
|
+
`[VllmProvider] Reasoning ceiling hit (${reasoningTokens} >= ${MAX_REASONING_TOKENS} est. tokens). Ending turn as length cutoff so the reasoning-cutoff continuation can land.`
|
|
597
|
+
);
|
|
598
|
+
break;
|
|
599
|
+
}
|
|
600
|
+
} catch (err) {
|
|
601
|
+
console.error(`[VllmProvider] Partial SSE parse anomaly (token undercount): ${err.message}`);
|
|
602
|
+
}
|
|
603
|
+
}
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
if (reasoningCeilingHit) break;
|
|
607
|
+
}
|
|
608
|
+
} finally {
|
|
609
|
+
disarmIdle();
|
|
610
|
+
if (cleanup) cleanup();
|
|
611
|
+
// Release the underlying connection when we ended the stream early
|
|
612
|
+
// (reasoning ceiling) so the engine sees a closed request promptly.
|
|
613
|
+
try {
|
|
614
|
+
await reader.cancel();
|
|
615
|
+
} catch {}
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
// Flush any held think-tag buffer at stream end: an unclosed think
|
|
619
|
+
// segment is reasoning; a partial opening tag is plain content.
|
|
620
|
+
flushThinkPending();
|
|
621
|
+
|
|
622
|
+
const totalMs = Math.max(1, Date.now() - t0);
|
|
623
|
+
const prefillMs = ttft ?? totalMs;
|
|
624
|
+
const generationMs = Math.max(0, totalMs - prefillMs);
|
|
625
|
+
|
|
626
|
+
// Prompt-token telemetry: prefer the engine-reported usage (authoritative)
|
|
627
|
+
// from the stream_options terminal chunk; when the engine did not emit it,
|
|
628
|
+
// fall back to the chars-based estimate already computed for the
|
|
629
|
+
// context-headroom clamp and mark the metric as estimated so downstream
|
|
630
|
+
// consumers never mistake a guess for a measurement.
|
|
631
|
+
const enginePromptTokens =
|
|
632
|
+
engineUsage && Number.isFinite(engineUsage.prompt_tokens)
|
|
633
|
+
? engineUsage.prompt_tokens
|
|
634
|
+
: null;
|
|
635
|
+
const promptTokens = enginePromptTokens ?? estimatedPromptTokens;
|
|
636
|
+
const promptTokensEstimated = enginePromptTokens === null;
|
|
637
|
+
|
|
638
|
+
// Completion tokens: the engine's usage is authoritative over the local
|
|
639
|
+
// per-delta count when present; reasoning accounting is left untouched.
|
|
640
|
+
const engineCompletionTokens =
|
|
641
|
+
engineUsage && Number.isFinite(engineUsage.completion_tokens)
|
|
642
|
+
? engineUsage.completion_tokens
|
|
643
|
+
: null;
|
|
644
|
+
const effectiveCompletionTokens =
|
|
645
|
+
engineCompletionTokens ?? completionTokens;
|
|
646
|
+
|
|
647
|
+
// Industry-standard throughput rates & per-token latency, derived from the
|
|
648
|
+
// raw timestamps above. All are null when the denominator is non-positive
|
|
649
|
+
// (or, for TPOT, when there is not enough output to divide by) so consumers
|
|
650
|
+
// never see a fabricated rate.
|
|
651
|
+
// prefillTps: prompt tokens / prefill seconds (prefill throughput)
|
|
652
|
+
// decodeTps: completion tokens / decode seconds (decode throughput)
|
|
653
|
+
// tpotMs: vLLM Time-Per-Output-Token = decode time / (output tokens - 1)
|
|
654
|
+
const prefillTps =
|
|
655
|
+
prefillMs > 0 ? Number((promptTokens / (prefillMs / 1000)).toFixed(2)) : null;
|
|
656
|
+
const decodeTps =
|
|
657
|
+
generationMs > 0
|
|
658
|
+
? Number((effectiveCompletionTokens / (generationMs / 1000)).toFixed(2))
|
|
659
|
+
: null;
|
|
660
|
+
const tpotMs =
|
|
661
|
+
effectiveCompletionTokens > 1 && generationMs > 0
|
|
662
|
+
? Number((generationMs / (effectiveCompletionTokens - 1)).toFixed(2))
|
|
663
|
+
: null;
|
|
664
|
+
|
|
665
|
+
const metrics = {
|
|
666
|
+
promptTokens,
|
|
667
|
+
completionTokens: effectiveCompletionTokens,
|
|
668
|
+
ttftMs: prefillMs,
|
|
669
|
+
prefillMs,
|
|
670
|
+
generationMs,
|
|
671
|
+
totalMs,
|
|
672
|
+
reasoningTokens,
|
|
673
|
+
hadReasoning,
|
|
674
|
+
reasoningCeilingHit,
|
|
675
|
+
// Report the actual idle window that armed for this turn (deep vs
|
|
676
|
+
// shallow tier), not the static shallow default.
|
|
677
|
+
streamIdleTimeoutMs: idleTimeoutMs,
|
|
678
|
+
streamIdleTier: idleTier,
|
|
679
|
+
prefillTps,
|
|
680
|
+
decodeTps,
|
|
681
|
+
tpotMs,
|
|
682
|
+
...(promptTokensEstimated ? { promptTokensEstimated: true } : {}),
|
|
683
|
+
};
|
|
684
|
+
|
|
685
|
+
if (onMetrics) onMetrics(metrics);
|
|
686
|
+
|
|
687
|
+
const toolCalls = Array.from(toolCallsMap.values()).map((tc) => ({
|
|
688
|
+
id: tc.id,
|
|
689
|
+
type: "function",
|
|
690
|
+
function: {
|
|
691
|
+
name: tc.name,
|
|
692
|
+
arguments: tc.arguments,
|
|
693
|
+
},
|
|
694
|
+
}));
|
|
695
|
+
|
|
696
|
+
// Never synthesize a finish reason the engine never sent. A stream that
|
|
697
|
+
// ended with no reason and no real output stays null so the runner
|
|
698
|
+
// classifies it (retry → engine_empty_response) instead of mistaking a
|
|
699
|
+
// dead stream for a clean stop. The tool_calls fallback is retained for
|
|
700
|
+
// engines that legitimately end the stream after complete tool calls.
|
|
701
|
+
let effectiveFinish = finishReason;
|
|
702
|
+
if (!effectiveFinish) {
|
|
703
|
+
if (toolCalls.length > 0) {
|
|
704
|
+
effectiveFinish = "tool_calls";
|
|
705
|
+
} else if (fullContent.length > 0) {
|
|
706
|
+
effectiveFinish = "stop";
|
|
707
|
+
} // else: null — dead/empty stream, honestly reported
|
|
708
|
+
}
|
|
709
|
+
|
|
710
|
+
return {
|
|
711
|
+
content: fullContent,
|
|
712
|
+
reasoning: fullReasoning,
|
|
713
|
+
toolCalls,
|
|
714
|
+
finishReason: effectiveFinish,
|
|
715
|
+
metrics,
|
|
716
|
+
reasoningTokens,
|
|
717
|
+
hadReasoning,
|
|
718
|
+
};
|
|
719
|
+
}
|
|
720
|
+
}
|
|
721
|
+
|
|
722
|
+
/**
|
|
723
|
+
* Castor Plugin to mount VllmProviderService into Context.
|
|
724
|
+
*/
|
|
725
|
+
export function vllmProviderPlugin(ctx, options = {}) {
|
|
726
|
+
const provider = new VllmProviderService(options);
|
|
727
|
+
return ctx.provide("llm", provider);
|
|
728
|
+
}
|