mcp-castor 2026.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +487 -0
  2. package/bin/castor.js +706 -0
  3. package/index.js +206 -0
  4. package/package.json +97 -0
  5. package/skills/canary-test-staging/SKILL.md +24 -0
  6. package/skills/evo-mutation-rollback/SKILL.md +29 -0
  7. package/skills/hypothesis-generation/SKILL.md +26 -0
  8. package/skills/traceback-condensing/SKILL.md +26 -0
  9. package/src/castor_runner.js +469 -0
  10. package/src/config.js +1204 -0
  11. package/src/env.js +10 -0
  12. package/src/evo_engine.js +214 -0
  13. package/src/harness/core/events.js +75 -0
  14. package/src/harness/core/kernel.js +209 -0
  15. package/src/harness/evo/evaluator.js +156 -0
  16. package/src/harness/evo/evo_operator.js +550 -0
  17. package/src/harness/evo/lineage_dag.js +383 -0
  18. package/src/harness/evo/trace_repair.js +173 -0
  19. package/src/harness/evo/watchdog.js +72 -0
  20. package/src/harness/loop_detector.js +135 -0
  21. package/src/harness/runner.js +1216 -0
  22. package/src/harness/services/ast_service.js +1813 -0
  23. package/src/harness/services/event_logger.js +275 -0
  24. package/src/harness/services/mcp_bridge.js +408 -0
  25. package/src/harness/services/provider_vllm.js +728 -0
  26. package/src/harness/services/sandbox_fs.js +1238 -0
  27. package/src/harness/services/searxng_lifecycle.js +254 -0
  28. package/src/harness/services/shell_executor.js +264 -0
  29. package/src/harness/services/shell_validator.js +506 -0
  30. package/src/harness/services/web_service.js +828 -0
  31. package/src/platform.js +344 -0
  32. package/src/repetition_detector.js +139 -0
  33. package/src/semaphore.js +373 -0
  34. package/src/server_lifecycle.js +781 -0
  35. package/src/skills.js +400 -0
  36. package/src/state_pruner.js +392 -0
  37. package/src/task_registry.js +1357 -0
  38. package/src/telemetry.js +638 -0
  39. package/src/tools.js +997 -0
  40. package/src/wsl_bridge.js +629 -0
  41. package/src/wsl_env.js +171 -0
  42. package/stream_proxy.js +453 -0
@@ -0,0 +1,728 @@
1
+ /**
2
+ * Direct vLLM HTTP/SSE Streaming Client (Castor vLLM Provider)
3
+ *
4
+ * Provides:
5
+ * - Direct HTTP streaming against vLLM (:18020) or Stream Proxy (:18022)
6
+ * - Proactive keep-alive frame handling
7
+ * - OpenAI-compatible function/tool calling parser
8
+ * - Live token velocity (tokens/sec) and TTFT measurement
9
+ * - Reversible Castor plugin binding
10
+ */
11
+
12
+ import {
13
+ VLLM_PORT,
14
+ MAX_TOKENS,
15
+ MAX_LEN_HUGE,
16
+ getReasoningEffort,
17
+ STREAM_IDLE_TIMEOUT_MS,
18
+ STREAM_IDLE_TIMEOUT_MS_DEEP,
19
+ STREAM_IDLE_DEPTH_TOKENS,
20
+ MAX_REASONING_TOKENS,
21
+ MODEL,
22
+ BASE_URL,
23
+ STREAM_PROXY_PORT,
24
+ } from "../../config.js";
25
+
26
+ export class VllmProviderService {
27
+ constructor(options = {}) {
28
+ // Engine identity is sourced from config.js (env > ~/.castor/config.json >
29
+ // built-in defaults) so the provider never hardcodes a model name or
30
+ // endpoint that could drift from the rest of the harness.
31
+ this.baseUrl =
32
+ options.baseUrl ||
33
+ (BASE_URL === `http://localhost:${VLLM_PORT}/v1`
34
+ ? `http://127.0.0.1:${STREAM_PROXY_PORT}/v1`
35
+ : BASE_URL);
36
+ this.fallbackUrl = options.fallbackUrl || `http://127.0.0.1:${VLLM_PORT}/v1`;
37
+ const rawModel = options.model || MODEL;
38
+ this.model =
39
+ rawModel && rawModel.toLowerCase() === "qwen3.8-27b"
40
+ ? "qwen3.8-27b"
41
+ : rawModel;
42
+ this.defaultTemperature = options.temperature ?? 0.0;
43
+ this.defaultMaxTokens = options.maxTokens ?? MAX_TOKENS;
44
+ }
45
+
46
+ /**
47
+ * Fetches available models from the vLLM server.
48
+ *
49
+ * Every failure path (network error, non-2xx, unparseable body, missing
50
+ * `data` array) throws an error carrying the upstream status/detail; only a
51
+ * genuine 2xx with a parseable `data` array yields a model list.
52
+ * @returns {Promise<string[]>} The list of model ids.
53
+ * @throws {Error} When the probe fails for any reason.
54
+ */
55
+ async listModels() {
56
+ let res;
57
+ try {
58
+ res = await fetch(`${this.baseUrl}/models`, {
59
+ signal: AbortSignal.timeout(5000),
60
+ });
61
+ } catch (err) {
62
+ // Network-level failure (connection refused / timeout / DNS): the
63
+ // engine (or proxy) is unreachable. Surface it verbatim — never a
64
+ // fabricated list.
65
+ throw new Error(
66
+ `listModels: vLLM /models probe failed against ${this.baseUrl}: ${
67
+ err && err.message ? err.message : String(err)
68
+ }`
69
+ );
70
+ }
71
+ if (!res.ok) {
72
+ // Non-2xx: the engine answered but rejected the probe. Carry the
73
+ // upstream status + body so the failure is decidable.
74
+ let detail = "";
75
+ try {
76
+ detail = await res.text();
77
+ } catch {}
78
+ throw new Error(
79
+ `listModels: vLLM /models probe returned HTTP ${res.status} ${res.statusText} from ${this.baseUrl}: ${detail}`
80
+ );
81
+ }
82
+ let data;
83
+ try {
84
+ data = await res.json();
85
+ } catch (err) {
86
+ throw new Error(
87
+ `listModels: vLLM /models returned 200 but an unparseable body from ${this.baseUrl}: ${
88
+ err && err.message ? err.message : String(err)
89
+ }`
90
+ );
91
+ }
92
+ if (!data || !Array.isArray(data.data)) {
93
+ throw new Error(
94
+ `listModels: vLLM /models returned 200 but no 'data' array from ${this.baseUrl} (body: ${JSON.stringify(data)})`
95
+ );
96
+ }
97
+ return data.data.map((m) => m.id);
98
+ }
99
+
100
+ /**
101
+ * Streams a chat completion turn from vLLM.
102
+ *
103
+ * @param {object} params
104
+ * @param {Array<object>} params.messages
105
+ * @param {Array<object>} [params.tools] OpenAI-formatted tools
106
+ * @param {number} [params.temperature]
107
+ * @param {number} [params.maxTokens]
108
+ * @param {string} [params.reasoningEffort] Task-local reasoning-effort tier
109
+ * (one of REASONING_EFFORT_TIERS). When provided it overrides the
110
+ * QWEN_REASONING_EFFORT env default for THIS request only — no
111
+ * process.env mutation, so it never leaks across concurrent tasks.
112
+ * @param {AbortSignal} [params.signal]
113
+ * @param {(token: string) => void} [params.onToken]
114
+ * @param {(metric: object) => void} [params.onMetrics]
115
+ * @returns {Promise<{
116
+ * content: string,
117
+ * toolCalls: Array<{ id: string, name: string, arguments: string }>,
118
+ * finishReason: string | null,
119
+ * metrics: { promptTokens: number, completionTokens: number, ttftMs: number, prefillMs: number, generationMs: number, totalMs: number, reasoningTokens: number, hadReasoning: boolean, reasoningCeilingHit: boolean, streamIdleTimeoutMs: number, streamIdleTier: "shallow" | "deep", promptTokensEstimated?: boolean }
120
+ * }>}
121
+ */
122
+ async streamChat({
123
+ messages,
124
+ tools = [],
125
+ temperature = this.defaultTemperature,
126
+ maxTokens = this.defaultMaxTokens,
127
+ reasoningEffort,
128
+ signal,
129
+ sessionId,
130
+ onToken,
131
+ onMetrics,
132
+ }) {
133
+ const t0 = Date.now();
134
+
135
+ // Dynamic headroom clamping against MAX_LEN_HUGE (245,760)
136
+ const promptChars = JSON.stringify(messages).length + (tools && tools.length > 0 ? JSON.stringify(tools).length : 0);
137
+ const estimatedPromptTokens = Math.ceil(promptChars / 3.5);
138
+ const maxPossibleHeadroom = Math.max(0, MAX_LEN_HUGE - estimatedPromptTokens - 128);
139
+
140
+ if (maxPossibleHeadroom < 1024) {
141
+ throw new Error(
142
+ `ContextExhaustedError: Prompt consumes ~${estimatedPromptTokens} tokens, leaving insufficient headroom (<1024) under model context limit (${MAX_LEN_HUGE}).`
143
+ );
144
+ }
145
+
146
+ const clampedMaxTokens = Math.min(maxTokens, maxPossibleHeadroom);
147
+
148
+ const payload = {
149
+ model: this.model,
150
+ messages,
151
+ stream: true,
152
+ temperature,
153
+ max_tokens: clampedMaxTokens,
154
+ // vLLM 0.28+: request the engine-reported token usage in a terminal
155
+ // SSE chunk (empty choices + a `usage` field). Additive only — engines
156
+ // that ignore it simply never emit the chunk, and we fall back to the
157
+ // chars-based estimate below.
158
+ stream_options: { include_usage: true },
159
+ };
160
+
161
+ if (tools && tools.length > 0) {
162
+ payload.tools = tools;
163
+ payload.tool_choice = "auto";
164
+ }
165
+
166
+ // Reasoning-effort passthrough: a task-local `reasoningEffort` param
167
+ // (threaded from the qwen_coworker dispatch) takes precedence; when it is
168
+ // absent the existing dynamic QWEN_REASONING_EFFORT read is the fallback
169
+ // (unchanged behavior). Forwarded to the vLLM chat template so the engine
170
+ // can trade thinking depth for latency per dispatch. When neither is set,
171
+ // send nothing and let the server default apply.
172
+ const effectiveReasoningEffort = reasoningEffort || getReasoningEffort();
173
+ if (effectiveReasoningEffort) {
174
+ payload.chat_template_kwargs = {
175
+ reasoning_effort: effectiveReasoningEffort,
176
+ };
177
+ }
178
+
179
+ let activeUrl = this.baseUrl;
180
+ let response;
181
+
182
+ // Compose an internal abort controller with the caller's signal so the
183
+ // stream-idle watchdog and the reasoning ceiling can end the request
184
+ // themselves, while an external cancel still propagates.
185
+ const streamController = new AbortController();
186
+ const onExternalAbort = () => streamController.abort(signal?.reason);
187
+ if (signal) {
188
+ if (signal.aborted) streamController.abort(signal.reason);
189
+ else signal.addEventListener("abort", onExternalAbort, { once: true });
190
+ }
191
+
192
+ try {
193
+ try {
194
+ response = await fetch(`${activeUrl}/chat/completions`, {
195
+ method: "POST",
196
+ headers: { "Content-Type": "application/json" },
197
+ body: JSON.stringify(payload),
198
+ signal: streamController.signal,
199
+ });
200
+ } catch (err) {
201
+ // Fallback directly to upstream vLLM port if proxy connection refused.
202
+ // An abort that ORIGINATED from the caller (not a proxy connect
203
+ // failure) must NOT be retried against the fallback.
204
+ if (signal?.aborted) throw err;
205
+ activeUrl = this.fallbackUrl;
206
+ response = await fetch(`${activeUrl}/chat/completions`, {
207
+ method: "POST",
208
+ headers: { "Content-Type": "application/json" },
209
+ body: JSON.stringify(payload),
210
+ signal: streamController.signal,
211
+ });
212
+ }
213
+
214
+ if (!response.ok) {
215
+ const errText = await response.text().catch(() => "");
216
+ throw new Error(`vLLM stream error (${response.status} ${response.statusText}): ${errText}`);
217
+ }
218
+
219
+ return await this._consumeStream({
220
+ response,
221
+ decoder: new TextDecoder("utf-8"),
222
+ t0,
223
+ // Threaded from streamChat (where the chars-based estimate is computed
224
+ // for the context-headroom clamp) so _consumeStream can fall back to it
225
+ // when the engine does not emit a stream_options usage chunk.
226
+ estimatedPromptTokens,
227
+ sessionId,
228
+ onToken,
229
+ onMetrics,
230
+ controller: streamController,
231
+ cleanup: () => {
232
+ if (signal) signal.removeEventListener("abort", onExternalAbort);
233
+ },
234
+ });
235
+ } finally {
236
+ if (signal) signal.removeEventListener("abort", onExternalAbort);
237
+ }
238
+ }
239
+
240
+ /**
241
+ * Reads the SSE body of an established streaming response.
242
+ *
243
+ * - Stream-idle watchdog: if no meaningful SSE frame arrives within the
244
+ * armed idle window, the request is aborted and the turn fails. The
245
+ * window is STREAM_IDLE_TIMEOUT_MS (shallow) for normal turns, or
246
+ * STREAM_IDLE_TIMEOUT_MS_DEEP when the estimated prompt tokens reach
247
+ * STREAM_IDLE_DEPTH_TOKENS (deep-context turns). The fired tier is named
248
+ * in the thrown error and in metrics.streamIdleTier. Proxy keep-alive
249
+ * comment frames (": keep-alive") do not reset the watchdog.
250
+ * - Reasoning ceiling: when estimated reasoning tokens exceed
251
+ * MAX_REASONING_TOKENS in a single turn, the stream is ended locally with
252
+ * finish_reason "length" and reasoningCeilingHit set.
253
+ * - finish_reason: a stream that ends with no finish_reason and produced
254
+ * neither content nor tool calls is reported as null, not synthesized as
255
+ * "stop".
256
+ *
257
+ * @param {object} params
258
+ * @param {Response} params.response
259
+ * @param {TextDecoder} params.decoder
260
+ * @param {number} params.t0
261
+ * @param {number} params.estimatedPromptTokens
262
+ * @param {string} [params.sessionId]
263
+ * @param {(token: string) => void} [params.onToken]
264
+ * @param {(metric: object) => void} [params.onMetrics]
265
+ * @param {AbortController} params.controller
266
+ * @param {() => void} [params.cleanup]
267
+ * @returns {Promise<{
268
+ * content: string,
269
+ * reasoning: string,
270
+ * toolCalls: Array<{ id: string, name: string, arguments: string }>,
271
+ * finishReason: string | null,
272
+ * metrics: object,
273
+ * reasoningTokens: number,
274
+ * hadReasoning: boolean
275
+ * }>}
276
+ */
277
+ async _consumeStream({
278
+ response,
279
+ decoder,
280
+ t0,
281
+ estimatedPromptTokens,
282
+ sessionId,
283
+ onToken,
284
+ onMetrics,
285
+ controller,
286
+ cleanup,
287
+ }) {
288
+ const reader = response.body.getReader();
289
+ let buffer = "";
290
+ let fullContent = "";
291
+ let fullReasoning = "";
292
+ let finishReason = null;
293
+ let ttft = null;
294
+ let completionTokens = 0;
295
+ let reasoningTokens = 0;
296
+ let hadReasoning = false;
297
+ let reasoningCeilingHit = false;
298
+ let idleTimedOut = false;
299
+ let idleTimer = null;
300
+ // Engine-reported usage from the terminal stream_options chunk (vLLM 0.28+).
301
+ // Null until the engine emits it; the chars-based estimate is the fallback.
302
+ let engineUsage = null;
303
+ const toolCallsMap = new Map(); // index -> { id, name, arguments }
304
+
305
+ // ------------------------------------------------------------------
306
+ // Inline think-tag normalization (Ollama / llama-server style):
307
+ // engines without a reasoning parser emit thinking segments inside
308
+ // delta.content wrapped in backtick-think tags. A small per-stream
309
+ // state machine splits the stream so the thinking lands in
310
+ // fullReasoning (counted, never surfaced to the user) and only the
311
+ // stripped text reaches fullContent / onToken. The pending buffer
312
+ // holds text whose tag boundary is not yet resolvable (a tag split
313
+ // across chunk boundaries).
314
+ // ------------------------------------------------------------------
315
+ const THINK_OPEN = String.fromCharCode(96) + "think";
316
+ const THINK_CLOSE = String.fromCharCode(96) + "think" + String.fromCharCode(96);
317
+ let thinkPending = "";
318
+ let inThink = false;
319
+
320
+ const flushThinkPending = () => {
321
+ if (!thinkPending) return;
322
+ if (inThink) {
323
+ // Unclosed think segment at stream end: the remainder is reasoning.
324
+ fullReasoning += thinkPending;
325
+ reasoningTokens += Math.max(1, Math.round(thinkPending.length / 4));
326
+ hadReasoning = true;
327
+ } else {
328
+ // A partial opening tag that never completed is plain content.
329
+ fullContent += thinkPending;
330
+ if (onToken) onToken(thinkPending);
331
+ }
332
+ thinkPending = "";
333
+ };
334
+
335
+ const processThinkDelta = (text) => {
336
+ thinkPending += text;
337
+ for (;;) {
338
+ if (inThink) {
339
+ const closeIdx = thinkPending.indexOf(THINK_CLOSE);
340
+ if (closeIdx < 0) return; // hold until the closing tag arrives
341
+ const thinking = thinkPending.slice(0, closeIdx);
342
+ if (thinking.length > 0) {
343
+ fullReasoning += thinking;
344
+ reasoningTokens += Math.max(1, Math.round(thinking.length / 4));
345
+ hadReasoning = true;
346
+ }
347
+ thinkPending = thinkPending.slice(closeIdx + THINK_CLOSE.length);
348
+ inThink = false;
349
+ continue;
350
+ }
351
+ const openIdx = thinkPending.indexOf(THINK_OPEN);
352
+ if (openIdx < 0) {
353
+ // Hold a trailing partial opening tag (e.g., a backtick-th
354
+ // split across chunks) so it can be matched once the rest arrives.
355
+ let hold = 0;
356
+ for (
357
+ let k = Math.min(THINK_OPEN.length, thinkPending.length);
358
+ k > 0;
359
+ k--
360
+ ) {
361
+ if (thinkPending.endsWith(THINK_OPEN.slice(0, k))) {
362
+ hold = k;
363
+ break;
364
+ }
365
+ }
366
+ const emit = thinkPending.slice(0, thinkPending.length - hold);
367
+ if (emit) {
368
+ fullContent += emit;
369
+ if (onToken) onToken(emit);
370
+ }
371
+ thinkPending = thinkPending.slice(thinkPending.length - hold);
372
+ return;
373
+ }
374
+ const closeIdx = thinkPending.indexOf(
375
+ THINK_CLOSE,
376
+ openIdx + THINK_OPEN.length
377
+ );
378
+ const before = thinkPending.slice(0, openIdx);
379
+ if (before) {
380
+ fullContent += before;
381
+ if (onToken) onToken(before);
382
+ }
383
+ if (closeIdx < 0) {
384
+ // Opening tag present but not yet closed: enter think mode and
385
+ // hold the remainder until the closing tag arrives.
386
+ inThink = true;
387
+ thinkPending = thinkPending.slice(openIdx + THINK_OPEN.length);
388
+ return;
389
+ }
390
+ const thinking = thinkPending.slice(
391
+ openIdx + THINK_OPEN.length,
392
+ closeIdx
393
+ );
394
+ if (thinking.length > 0) {
395
+ fullReasoning += thinking;
396
+ reasoningTokens += Math.max(1, Math.round(thinking.length / 4));
397
+ hadReasoning = true;
398
+ }
399
+ thinkPending = thinkPending.slice(closeIdx + THINK_CLOSE.length);
400
+ }
401
+ };
402
+
403
+ // Depth-aware idle tier: a deep-context prompt (estimated prompt tokens
404
+ // >= STREAM_IDLE_DEPTH_TOKENS) gets the longer DEEP window; normal turns
405
+ // keep the SHALLOW window. The tier is chosen from the chars-based
406
+ // estimate already computed in streamChat.
407
+ const isDeep = estimatedPromptTokens >= STREAM_IDLE_DEPTH_TOKENS;
408
+ const idleTimeoutMs = isDeep ? STREAM_IDLE_TIMEOUT_MS_DEEP : STREAM_IDLE_TIMEOUT_MS;
409
+ const idleTier = isDeep ? "deep" : "shallow";
410
+ let deepTierLogged = false;
411
+
412
+ const disarmIdle = () => {
413
+ if (idleTimer) {
414
+ clearTimeout(idleTimer);
415
+ idleTimer = null;
416
+ }
417
+ };
418
+ const armIdle = () => {
419
+ disarmIdle();
420
+ idleTimer = setTimeout(() => {
421
+ idleTimedOut = true;
422
+ controller.abort(new Error(`stream idle > ${idleTimeoutMs}ms (${idleTier} tier)`));
423
+ }, idleTimeoutMs);
424
+ if (typeof idleTimer.unref === "function") idleTimer.unref();
425
+ // Log once when the deep tier arms.
426
+ if (isDeep && !deepTierLogged) {
427
+ deepTierLogged = true;
428
+ console.error(
429
+ `[VllmProvider] Deep-context turn (est. prompt ${estimatedPromptTokens} tokens >= ${STREAM_IDLE_DEPTH_TOKENS}): arming ${idleTimeoutMs}ms (${idleTier} tier) stream-idle watchdog.`
430
+ );
431
+ }
432
+ };
433
+
434
+ let currentSseEvent = null;
435
+ try {
436
+ armIdle();
437
+ while (true) {
438
+ let readResult;
439
+ try {
440
+ readResult = await reader.read();
441
+ } catch (err) {
442
+ if (idleTimedOut) {
443
+ const sid = sessionId || "SESSION_ID";
444
+ const inspectCmd = `node mcp-castor/src/harness/services/event_logger.js ${sid} 5`;
445
+ throw new Error(
446
+ `vLLM stream idle timeout (${idleTier} tier, ${idleTimeoutMs}ms): no meaningful SSE tokens emitted within window.\n` +
447
+ `[Orchestrator Advisory]: There might have been an issue (e.g. extended GPU contention or deliberation).\n` +
448
+ `You can run:\n` +
449
+ ` ${inspectCmd}\n` +
450
+ `to inspect the last few traces, and decide whether to roll into a new session (e.g. '${sid}_stage2'), or use the same session and same effort to finish/continue.`
451
+ );
452
+ }
453
+ throw err;
454
+ }
455
+ const { done, value } = readResult;
456
+ if (done) break;
457
+
458
+ buffer += decoder.decode(value, { stream: true });
459
+ const lines = buffer.split("\n");
460
+ buffer = lines.pop(); // keep last incomplete line
461
+
462
+ for (const line of lines) {
463
+ const trimmed = line.trim();
464
+ if (!trimmed || trimmed.startsWith(":")) {
465
+ // SSE comment / keep-alive heartbeat frame — NOT engine activity.
466
+ continue;
467
+ }
468
+
469
+ // Any meaningful SSE frame proves the engine (or an interceptor)
470
+ // is alive and producing; reset the idle watchdog.
471
+ armIdle();
472
+
473
+ if (trimmed.startsWith("event: ")) {
474
+ currentSseEvent = trimmed.slice(7).trim();
475
+ continue;
476
+ }
477
+
478
+ if (trimmed === "data: [DONE]") {
479
+ break;
480
+ }
481
+
482
+ if (trimmed.startsWith("data: ")) {
483
+ const jsonStr = trimmed.slice(6);
484
+ if (currentSseEvent === "error") {
485
+ let errMsg = jsonStr;
486
+ try {
487
+ const parsedErr = JSON.parse(jsonStr);
488
+ errMsg = parsedErr.error?.message || parsedErr.message || jsonStr;
489
+ } catch {}
490
+ throw new Error(`vLLM upstream SSE error: ${errMsg}`);
491
+ }
492
+ currentSseEvent = null;
493
+
494
+ try {
495
+ const chunk = JSON.parse(jsonStr);
496
+ if (chunk.error && !chunk.choices) {
497
+ const errMsg = chunk.error.message || JSON.stringify(chunk.error);
498
+ throw new Error(`vLLM upstream error: ${errMsg}`);
499
+ }
500
+ if (chunk.id === "chatcmpl-stream-err") {
501
+ const errMsg = chunk.choices?.[0]?.delta?.content || "vLLM stream error";
502
+ throw new Error(`vLLM stream error: ${errMsg}`);
503
+ }
504
+
505
+ // stream_options { include_usage: true } (vLLM 0.28+): the engine
506
+ // emits a terminal chunk whose `choices` is EMPTY and whose
507
+ // `usage` field carries the authoritative prompt/completion token
508
+ // counts. Capture it here, before the `choice` guard below, so the
509
+ // empty-choices chunk is consumed gracefully and does not
510
+ // contribute to the finish-reason or content signals.
511
+ if (chunk.usage && typeof chunk.usage === "object") {
512
+ engineUsage = chunk.usage;
513
+ }
514
+
515
+ const choice = chunk.choices?.[0];
516
+ if (!choice) continue;
517
+
518
+ if (choice.finish_reason) {
519
+ finishReason = choice.finish_reason;
520
+ }
521
+
522
+ const delta = choice.delta;
523
+ if (!delta) continue;
524
+
525
+ // Server-side reasoning (thinking) is streamed by vLLM's
526
+ // --reasoning-parser qwen3 as delta.reasoning, while some engines
527
+ // / older parsers use delta.reasoning_content. Normalize both so
528
+ // accounting works regardless of which the engine emits. It is
529
+ // accounted for (so the ledger shows thinking volume and TTFT
530
+ // reflects the first output of any kind) but is not appended to
531
+ // fullContent or forwarded to onToken — the user-visible token
532
+ // stream stays clean, and the engine-side /metrics token counters
533
+ // already cover watchdog velocity.
534
+ const reasoningText =
535
+ typeof delta.reasoning === "string"
536
+ ? delta.reasoning
537
+ : typeof delta.reasoning_content === "string"
538
+ ? delta.reasoning_content
539
+ : null;
540
+ if (reasoningText && reasoningText.length > 0) {
541
+ hadReasoning = true;
542
+ fullReasoning += reasoningText;
543
+ // ~chars/4 is a reasonable token estimate for thinking text.
544
+ reasoningTokens += Math.max(1, Math.round(reasoningText.length / 4));
545
+ }
546
+
547
+ // Measure Time to First Token: the first delta of ANY kind
548
+ // (content, tool call, or reasoning) means generation started.
549
+ if (
550
+ ttft === null &&
551
+ (delta.content || delta.tool_calls || reasoningText)
552
+ ) {
553
+ ttft = Date.now() - t0;
554
+ }
555
+
556
+ // Stream text content. Engines without a reasoning parser
557
+ // (Ollama, llama-server) emit think-tag segments inside
558
+ // delta.content; the state machine above splits them so the
559
+ // thinking lands in fullReasoning and only the stripped text
560
+ // reaches fullContent / onToken.
561
+ if (delta.content) {
562
+ completionTokens++;
563
+ processThinkDelta(delta.content);
564
+ }
565
+
566
+ // Stream tool calls
567
+ if (delta.tool_calls && Array.isArray(delta.tool_calls)) {
568
+ for (const tc of delta.tool_calls) {
569
+ const idx = tc.index ?? 0;
570
+ if (!toolCallsMap.has(idx)) {
571
+ toolCallsMap.set(idx, {
572
+ id: tc.id || `call_${Date.now()}_${idx}`,
573
+ name: tc.function?.name || "",
574
+ arguments: tc.function?.arguments || "",
575
+ });
576
+ } else {
577
+ const existing = toolCallsMap.get(idx);
578
+ if (tc.id && !existing.id) existing.id = tc.id;
579
+ if (tc.function?.name) existing.name += tc.function.name;
580
+ if (tc.function?.arguments) existing.arguments += tc.function.arguments;
581
+ }
582
+ }
583
+ }
584
+
585
+ // Reasoning-ceiling enforcement: cutting the stream here reports
586
+ // as a length cutoff (with hadReasoning already true), which
587
+ // routes the runner into the reasoning-cutoff continuation
588
+ // directive instead of a retry/failure path.
589
+ if (
590
+ !reasoningCeilingHit &&
591
+ reasoningTokens >= MAX_REASONING_TOKENS
592
+ ) {
593
+ reasoningCeilingHit = true;
594
+ finishReason = "length";
595
+ console.error(
596
+ `[VllmProvider] Reasoning ceiling hit (${reasoningTokens} >= ${MAX_REASONING_TOKENS} est. tokens). Ending turn as length cutoff so the reasoning-cutoff continuation can land.`
597
+ );
598
+ break;
599
+ }
600
+ } catch (err) {
601
+ console.error(`[VllmProvider] Partial SSE parse anomaly (token undercount): ${err.message}`);
602
+ }
603
+ }
604
+ }
605
+
606
+ if (reasoningCeilingHit) break;
607
+ }
608
+ } finally {
609
+ disarmIdle();
610
+ if (cleanup) cleanup();
611
+ // Release the underlying connection when we ended the stream early
612
+ // (reasoning ceiling) so the engine sees a closed request promptly.
613
+ try {
614
+ await reader.cancel();
615
+ } catch {}
616
+ }
617
+
618
+ // Flush any held think-tag buffer at stream end: an unclosed think
619
+ // segment is reasoning; a partial opening tag is plain content.
620
+ flushThinkPending();
621
+
622
+ const totalMs = Math.max(1, Date.now() - t0);
623
+ const prefillMs = ttft ?? totalMs;
624
+ const generationMs = Math.max(0, totalMs - prefillMs);
625
+
626
+ // Prompt-token telemetry: prefer the engine-reported usage (authoritative)
627
+ // from the stream_options terminal chunk; when the engine did not emit it,
628
+ // fall back to the chars-based estimate already computed for the
629
+ // context-headroom clamp and mark the metric as estimated so downstream
630
+ // consumers never mistake a guess for a measurement.
631
+ const enginePromptTokens =
632
+ engineUsage && Number.isFinite(engineUsage.prompt_tokens)
633
+ ? engineUsage.prompt_tokens
634
+ : null;
635
+ const promptTokens = enginePromptTokens ?? estimatedPromptTokens;
636
+ const promptTokensEstimated = enginePromptTokens === null;
637
+
638
+ // Completion tokens: the engine's usage is authoritative over the local
639
+ // per-delta count when present; reasoning accounting is left untouched.
640
+ const engineCompletionTokens =
641
+ engineUsage && Number.isFinite(engineUsage.completion_tokens)
642
+ ? engineUsage.completion_tokens
643
+ : null;
644
+ const effectiveCompletionTokens =
645
+ engineCompletionTokens ?? completionTokens;
646
+
647
+ // Industry-standard throughput rates & per-token latency, derived from the
648
+ // raw timestamps above. All are null when the denominator is non-positive
649
+ // (or, for TPOT, when there is not enough output to divide by) so consumers
650
+ // never see a fabricated rate.
651
+ // prefillTps: prompt tokens / prefill seconds (prefill throughput)
652
+ // decodeTps: completion tokens / decode seconds (decode throughput)
653
+ // tpotMs: vLLM Time-Per-Output-Token = decode time / (output tokens - 1)
654
+ const prefillTps =
655
+ prefillMs > 0 ? Number((promptTokens / (prefillMs / 1000)).toFixed(2)) : null;
656
+ const decodeTps =
657
+ generationMs > 0
658
+ ? Number((effectiveCompletionTokens / (generationMs / 1000)).toFixed(2))
659
+ : null;
660
+ const tpotMs =
661
+ effectiveCompletionTokens > 1 && generationMs > 0
662
+ ? Number((generationMs / (effectiveCompletionTokens - 1)).toFixed(2))
663
+ : null;
664
+
665
+ const metrics = {
666
+ promptTokens,
667
+ completionTokens: effectiveCompletionTokens,
668
+ ttftMs: prefillMs,
669
+ prefillMs,
670
+ generationMs,
671
+ totalMs,
672
+ reasoningTokens,
673
+ hadReasoning,
674
+ reasoningCeilingHit,
675
+ // Report the actual idle window that armed for this turn (deep vs
676
+ // shallow tier), not the static shallow default.
677
+ streamIdleTimeoutMs: idleTimeoutMs,
678
+ streamIdleTier: idleTier,
679
+ prefillTps,
680
+ decodeTps,
681
+ tpotMs,
682
+ ...(promptTokensEstimated ? { promptTokensEstimated: true } : {}),
683
+ };
684
+
685
+ if (onMetrics) onMetrics(metrics);
686
+
687
+ const toolCalls = Array.from(toolCallsMap.values()).map((tc) => ({
688
+ id: tc.id,
689
+ type: "function",
690
+ function: {
691
+ name: tc.name,
692
+ arguments: tc.arguments,
693
+ },
694
+ }));
695
+
696
+ // Never synthesize a finish reason the engine never sent. A stream that
697
+ // ended with no reason and no real output stays null so the runner
698
+ // classifies it (retry → engine_empty_response) instead of mistaking a
699
+ // dead stream for a clean stop. The tool_calls fallback is retained for
700
+ // engines that legitimately end the stream after complete tool calls.
701
+ let effectiveFinish = finishReason;
702
+ if (!effectiveFinish) {
703
+ if (toolCalls.length > 0) {
704
+ effectiveFinish = "tool_calls";
705
+ } else if (fullContent.length > 0) {
706
+ effectiveFinish = "stop";
707
+ } // else: null — dead/empty stream, honestly reported
708
+ }
709
+
710
+ return {
711
+ content: fullContent,
712
+ reasoning: fullReasoning,
713
+ toolCalls,
714
+ finishReason: effectiveFinish,
715
+ metrics,
716
+ reasoningTokens,
717
+ hadReasoning,
718
+ };
719
+ }
720
+ }
721
+
722
+ /**
723
+ * Castor Plugin to mount VllmProviderService into Context.
724
+ */
725
+ export function vllmProviderPlugin(ctx, options = {}) {
726
+ const provider = new VllmProviderService(options);
727
+ return ctx.provide("llm", provider);
728
+ }