mcp-castor 2026.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +487 -0
  2. package/bin/castor.js +706 -0
  3. package/index.js +206 -0
  4. package/package.json +97 -0
  5. package/skills/canary-test-staging/SKILL.md +24 -0
  6. package/skills/evo-mutation-rollback/SKILL.md +29 -0
  7. package/skills/hypothesis-generation/SKILL.md +26 -0
  8. package/skills/traceback-condensing/SKILL.md +26 -0
  9. package/src/castor_runner.js +469 -0
  10. package/src/config.js +1204 -0
  11. package/src/env.js +10 -0
  12. package/src/evo_engine.js +214 -0
  13. package/src/harness/core/events.js +75 -0
  14. package/src/harness/core/kernel.js +209 -0
  15. package/src/harness/evo/evaluator.js +156 -0
  16. package/src/harness/evo/evo_operator.js +550 -0
  17. package/src/harness/evo/lineage_dag.js +383 -0
  18. package/src/harness/evo/trace_repair.js +173 -0
  19. package/src/harness/evo/watchdog.js +72 -0
  20. package/src/harness/loop_detector.js +135 -0
  21. package/src/harness/runner.js +1216 -0
  22. package/src/harness/services/ast_service.js +1813 -0
  23. package/src/harness/services/event_logger.js +275 -0
  24. package/src/harness/services/mcp_bridge.js +408 -0
  25. package/src/harness/services/provider_vllm.js +728 -0
  26. package/src/harness/services/sandbox_fs.js +1238 -0
  27. package/src/harness/services/searxng_lifecycle.js +254 -0
  28. package/src/harness/services/shell_executor.js +264 -0
  29. package/src/harness/services/shell_validator.js +506 -0
  30. package/src/harness/services/web_service.js +828 -0
  31. package/src/platform.js +344 -0
  32. package/src/repetition_detector.js +139 -0
  33. package/src/semaphore.js +373 -0
  34. package/src/server_lifecycle.js +781 -0
  35. package/src/skills.js +400 -0
  36. package/src/state_pruner.js +392 -0
  37. package/src/task_registry.js +1357 -0
  38. package/src/telemetry.js +638 -0
  39. package/src/tools.js +997 -0
  40. package/src/wsl_bridge.js +629 -0
  41. package/src/wsl_env.js +171 -0
  42. package/stream_proxy.js +453 -0
package/src/config.js ADDED
@@ -0,0 +1,1204 @@
1
+ import fs from "node:fs";
2
+ import os from "node:os";
3
+ import path from "node:path";
4
+ import { winHomeWsl } from "./wsl_env.js";
5
+ import { IS_WINDOWS } from "./env.js";
6
+
7
+ /**
8
+ * Platform detection flag for Windows environments.
9
+ * Re-exported from leaf env.js for backward compatibility.
10
+ * @type {boolean}
11
+ */
12
+ export { IS_WINDOWS };
13
+
14
+ /**
15
+ * Port for the vLLM OpenAI-compatible server.
16
+ * - Unit: port number
17
+ * - Default: 18020
18
+ * - Override: VLLM_PORT
19
+ * @type {number}
20
+ */
21
+ export const VLLM_PORT = parseInt(process.env.VLLM_PORT || "18020", 10);
22
+
23
+ /**
24
+ * Port for the zero-turn status and long-poll wait HTTP server.
25
+ * - Unit: port number
26
+ * - Default: 18021
27
+ * - Override: STATUS_PORT
28
+ * @type {number}
29
+ */
30
+ export const STATUS_PORT = parseInt(process.env.STATUS_PORT || "18021", 10);
31
+
32
+ /**
33
+ * Port for the local SSE stream sanitizer proxy.
34
+ * - Unit: port number
35
+ * - Default: 18022
36
+ * - Override: STREAM_PROXY_PORT (preferred) or VLLM_PROXY_PORT (legacy)
37
+ * - Config-file override: `stream_proxy_port` in ~/.castor/config.json
38
+ * (resolved in STREAM_PROXY_PORT_RESOLVED at the bottom of this file)
39
+ * @type {number}
40
+ */
41
+ export const STREAM_PROXY_PORT = parseInt(
42
+ process.env.STREAM_PROXY_PORT || process.env.VLLM_PROXY_PORT || "18022",
43
+ 10
44
+ );
45
+
46
+ /**
47
+ * Whether to use the stream proxy.
48
+ * - Default: true
49
+ * - Override: USE_STREAM_PROXY ("false" or "0" disables)
50
+ * - Config-file override: `use_stream_proxy` in ~/.castor/config.json
51
+ * (resolved in USE_STREAM_PROXY_RESOLVED at the bottom of this file)
52
+ * @type {boolean}
53
+ */
54
+ export const USE_STREAM_PROXY =
55
+ process.env.USE_STREAM_PROXY !== "false" && process.env.USE_STREAM_PROXY !== "0";
56
+
57
+ /**
58
+ * Base URL for vLLM API completions.
59
+ * @type {string}
60
+ */
61
+ export const BASE_URL = `http://localhost:${VLLM_PORT}/v1`;
62
+
63
+ /**
64
+ * Nominal maximum context length for the Qwen model architecture.
65
+ * - Unit: tokens
66
+ * - Value: 245760
67
+ * @type {number}
68
+ */
69
+ export const MAX_LEN_HUGE = 245760;
70
+ /**
71
+ * Maximum duration to wait for the vLLM engine to become healthy after boot.
72
+ * - Unit: milliseconds
73
+ * - Default: 480000 (8 minutes)
74
+ * @type {number}
75
+ */
76
+ export const BOOT_TIMEOUT_MS = 480_000;
77
+
78
+ /**
79
+ * Polling cadence when checking vLLM health during startup.
80
+ * - Unit: milliseconds
81
+ * - Default: 3000 (3 seconds)
82
+ * @type {number}
83
+ */
84
+ export const BOOT_POLL_MS = 3000;
85
+
86
+ /**
87
+ * Maximum completion tokens allowed for a single generation turn.
88
+ * - Unit: tokens
89
+ * - Default: 49152
90
+ * - Override: QWEN_MAX_TOKENS
91
+ * @type {number}
92
+ */
93
+ const DEFAULT_MAX_TOKENS = 49152;
94
+ export const MAX_TOKENS = (() => {
95
+ const parsed = parseInt(process.env.QWEN_MAX_TOKENS, 10);
96
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_MAX_TOKENS;
97
+ })();
98
+
99
+ /**
100
+ * Retrieves the current default reasoning effort for generation dispatches.
101
+ * - Override: QWEN_REASONING_EFFORT
102
+ * - Default: "medium"
103
+ * @returns {"xhigh"|"medium"|"low"}
104
+ */
105
+ export function getReasoningEffort() {
106
+ const v = process.env.QWEN_REASONING_EFFORT;
107
+ return v ? v : "medium";
108
+ }
109
+
110
+ /**
111
+ * Supported reasoning-effort tiers accepted by the engine's chat template.
112
+ * Canonical enum for tool schemas and dispatch validation.
113
+ * @type {readonly string[]}
114
+ */
115
+ export const REASONING_EFFORT_TIERS = ["xhigh", "medium", "low"];
116
+
117
+ /**
118
+ * Synchronous race window before yielding to asynchronous zero-turn HTTP wait.
119
+ * - Unit: milliseconds
120
+ * - Default: 15000 (15 seconds)
121
+ * - Override: QWEN_RACE_MS
122
+ * @type {number}
123
+ */
124
+ const DEFAULT_RACE_MS = 15_000;
125
+ export const RACE_MS = process.env.QWEN_RACE_MS
126
+ ? parseInt(process.env.QWEN_RACE_MS, 10)
127
+ : DEFAULT_RACE_MS;
128
+
129
+ /**
130
+ * Default maximum execution timeout for a Castor background task.
131
+ * - Unit: milliseconds
132
+ * - Default: 14400000 (4 hours)
133
+ * @type {number}
134
+ */
135
+ export const DEFAULT_TIMEOUT_MS = 14_400_000;
136
+
137
+ /**
138
+ * Enforced lower bound for task timeout configuration.
139
+ * - Unit: milliseconds
140
+ * - Default: 600000 (10 minutes)
141
+ * - Override: QWEN_MIN_TIMEOUT_MS
142
+ * @type {number}
143
+ */
144
+ const DEFAULT_MIN_TIMEOUT_MS = 600_000;
145
+ export const MIN_TIMEOUT_MS = (() => {
146
+ const parsed = parseInt(process.env.QWEN_MIN_TIMEOUT_MS, 10);
147
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_MIN_TIMEOUT_MS;
148
+ })();
149
+
150
+ /**
151
+ * Inactivity watchdog timeout for stream stalls or unhandled process locks.
152
+ * - Unit: milliseconds
153
+ * - Default: 1800000 (30 minutes)
154
+ * - Override: QWEN_INACTIVITY_TIMEOUT_MS
155
+ * @type {number}
156
+ */
157
+ export const INACTIVITY_TIMEOUT_MS = (() => {
158
+ const parsed = parseInt(process.env.QWEN_INACTIVITY_TIMEOUT_MS, 10);
159
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : 1_800_000;
160
+ })();
161
+
162
+ /**
163
+ * Timeout allowed before the initial streaming token is emitted by the engine.
164
+ * - Unit: milliseconds
165
+ * - Default: 240000 (4 minutes)
166
+ * - Override: QWEN_FIRST_TOKEN_TIMEOUT_MS
167
+ * @type {number}
168
+ */
169
+ export const FIRST_TOKEN_TIMEOUT_MS = (() => {
170
+ const parsed = parseInt(process.env.QWEN_FIRST_TOKEN_TIMEOUT_MS, 10);
171
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : 240_000;
172
+ })();
173
+
174
+ /**
175
+ * Maximum allowed idle duration between consecutive stream chunks before aborting.
176
+ * - Unit: milliseconds
177
+ * - Default: 1200000 (20 minutes)
178
+ * - Override: QWEN_STREAM_IDLE_TIMEOUT_MS
179
+ * @type {number}
180
+ */
181
+ const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 1_200_000; // 20 min
182
+ export const STREAM_IDLE_TIMEOUT_MS = (() => {
183
+ const parsed = parseInt(process.env.QWEN_STREAM_IDLE_TIMEOUT_MS, 10);
184
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_STREAM_IDLE_TIMEOUT_MS;
185
+ })();
186
+
187
+ /**
188
+ * Streaming idle timeout applied when prompt tokens exceed STREAM_IDLE_DEPTH_TOKENS.
189
+ * - Unit: milliseconds
190
+ * - Default: 2400000 (40 minutes)
191
+ * - Override: QWEN_STREAM_IDLE_TIMEOUT_DEEP_MS
192
+ * @type {number}
193
+ */
194
+ const DEFAULT_STREAM_IDLE_TIMEOUT_MS_DEEP = 2_400_000; // 40 min
195
+ export const STREAM_IDLE_TIMEOUT_MS_DEEP = (() => {
196
+ const parsed = parseInt(process.env.QWEN_STREAM_IDLE_TIMEOUT_DEEP_MS, 10);
197
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_STREAM_IDLE_TIMEOUT_MS_DEEP;
198
+ })();
199
+
200
+ /**
201
+ * Prompt token depth threshold that activates STREAM_IDLE_TIMEOUT_MS_DEEP.
202
+ * - Unit: tokens
203
+ * - Default: 35000
204
+ * - Override: QWEN_STREAM_IDLE_DEPTH_TOKENS
205
+ * @type {number}
206
+ */
207
+ const DEFAULT_STREAM_IDLE_DEPTH_TOKENS = 35_000;
208
+ export const STREAM_IDLE_DEPTH_TOKENS = (() => {
209
+ const parsed = parseInt(process.env.QWEN_STREAM_IDLE_DEPTH_TOKENS, 10);
210
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_STREAM_IDLE_DEPTH_TOKENS;
211
+ })();
212
+
213
+ /**
214
+ * Maximum reasoning tokens allowed per generation turn before terminating thinking.
215
+ * - Unit: tokens
216
+ * - Default: 32768
217
+ * - Override: QWEN_MAX_REASONING_TOKENS
218
+ * @type {number}
219
+ */
220
+ const DEFAULT_MAX_REASONING_TOKENS = 32_768;
221
+ export const MAX_REASONING_TOKENS = (() => {
222
+ const parsed = parseInt(process.env.QWEN_MAX_REASONING_TOKENS, 10);
223
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_MAX_REASONING_TOKENS;
224
+ })();
225
+
226
+ /**
227
+ * Additional execution time granted per supervisor lease extension.
228
+ * - Unit: milliseconds
229
+ * - Default: 600000 (10 minutes)
230
+ * @type {number}
231
+ */
232
+ export const EXTENSION_BONUS_TIMEOUT_MS = 600_000;
233
+
234
+ /**
235
+ * Minimum allowable task retention period (ms). Enforces that task JSON files
236
+ * outlive the maximum legal execution timeout plus inactivity watchdog margin.
237
+ * - Unit: milliseconds
238
+ * - Default: 16200000 (4h 30m)
239
+ * @type {number}
240
+ */
241
+ export const TASK_RETENTION_FLOOR_MS = DEFAULT_TIMEOUT_MS + 1_800_000;
242
+
243
+ /**
244
+ * Task artifact retention period on disk before automated pruning. Clamped to
245
+ * {@link TASK_RETENTION_FLOOR_MS} to prevent premature deletion of active tasks.
246
+ * - Unit: milliseconds
247
+ * - Default: 604800000 (7 days)
248
+ * - Override: QWEN_TASK_RETENTION_MS
249
+ * @type {number}
250
+ */
251
+ const DEFAULT_TASK_RETENTION_MS = 604_800_000;
252
+ export const TASK_RETENTION_MS = (() => {
253
+ const parsed = parseInt(process.env.QWEN_TASK_RETENTION_MS, 10);
254
+ const requested =
255
+ Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_TASK_RETENTION_MS;
256
+ return Math.max(requested, TASK_RETENTION_FLOOR_MS);
257
+ })();
258
+
259
+ /**
260
+ * Heartbeat staleness threshold for reaping orphaned tasks whose owner process is dead.
261
+ * - Unit: milliseconds
262
+ * - Default: 30000 (30 seconds)
263
+ * - Override: QWEN_ORPHAN_REAP_STALE_MS
264
+ * @type {number}
265
+ */
266
+ const DEFAULT_ORPHAN_REAP_STALE_MS = 30_000;
267
+ export const ORPHAN_REAP_STALE_MS = (() => {
268
+ const parsed = parseInt(process.env.QWEN_ORPHAN_REAP_STALE_MS, 10);
269
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_ORPHAN_REAP_STALE_MS;
270
+ })();
271
+
272
+ /**
273
+ * Maximum number of concurrent tasks executed in parallel.
274
+ * - Unit: count
275
+ * - Default: 1
276
+ * - Override: QWEN_MAX_CONCURRENT
277
+ * @type {number}
278
+ */
279
+ export const MAX_CONCURRENT_TASKS = process.env.QWEN_MAX_CONCURRENT
280
+ ? Math.max(1, parseInt(process.env.QWEN_MAX_CONCURRENT, 10))
281
+ : 1;
282
+
283
+ /**
284
+ * Heartbeat refresh interval for active task slot leases.
285
+ * - Unit: milliseconds
286
+ * - Default: 15000 (15 seconds)
287
+ * @type {number}
288
+ */
289
+ export const SLOT_HEARTBEAT_MS = 15_000;
290
+
291
+ /**
292
+ * Inactivity duration before an unrefreshed slot lease is declared wedged.
293
+ * - Unit: milliseconds
294
+ * - Default: 300000 (5 minutes)
295
+ * @type {number}
296
+ */
297
+ export const SLOT_WEDGED_MS = 300_000;
298
+
299
+ /**
300
+ * Polling cadence when waiting to acquire an exclusive slot lease.
301
+ * - Unit: milliseconds
302
+ * - Default: 1000 (1 second)
303
+ * @type {number}
304
+ */
305
+ export const SLOT_POLL_MS = 1_000;
306
+
307
+ /**
308
+ * Identifies whether execution is occurring within an automated test suite.
309
+ * @type {boolean}
310
+ */
311
+ export const IS_TEST_ENV = Boolean(
312
+ process.env.NODE_ENV === "test" ||
313
+ process.env.TEST_OFFLINE === "1" ||
314
+ (process.env.npm_lifecycle_event && process.env.npm_lifecycle_event.includes("test")) ||
315
+ process.argv.some((arg) => typeof arg === "string" && (arg.endsWith(".test.js") || arg.includes(".test.") || arg === "--test"))
316
+ );
317
+
318
+ if (IS_TEST_ENV && process.env.NODE_ENV !== "test") {
319
+ process.env.NODE_ENV = "test";
320
+ }
321
+
322
+ /**
323
+ * One-time migration from the legacy ~/.qwen state directory to ~/.castor.
324
+ * If the legacy directory exists and the new one does not, it is renamed
325
+ * (or copied as a fallback) so existing task state, config.json, and
326
+ * session ledgers carry over. Never throws; failures are logged to stderr.
327
+ *
328
+ * @param {string} legacyDir - legacy state dir (e.g. ~/.qwen)
329
+ * @param {string} newDir - new state dir (e.g. ~/.castor)
330
+ */
331
+ function migrateStateDir(legacyDir, newDir) {
332
+ try {
333
+ if (!fs.existsSync(legacyDir) || fs.existsSync(newDir)) return;
334
+ try {
335
+ fs.renameSync(legacyDir, newDir);
336
+ } catch {
337
+ // Cross-device or permission failure: fall back to a recursive copy.
338
+ fs.cpSync(legacyDir, newDir, { recursive: true });
339
+ }
340
+ process.stderr.write(
341
+ `[config] Migrated state directory ${legacyDir} -> ${newDir}\n`
342
+ );
343
+ } catch (err) {
344
+ process.stderr.write(
345
+ `[config] State dir migration failed (${err.message}); using ${newDir}\n`
346
+ );
347
+ }
348
+ }
349
+
350
+ /**
351
+ * Root directory for task state, slot leases, and session ledgers.
352
+ * In test environments, allocates an isolated temporary scratchpad.
353
+ * In production environments, resolves to ~/.castor (or WSL host equivalent),
354
+ * with a one-time migration from the legacy ~/.qwen directory.
355
+ * - Override: QWEN_STATE_DIR
356
+ * @type {string}
357
+ */
358
+ export const QWEN_STATE_DIR = process.env.QWEN_STATE_DIR || (() => {
359
+ if (IS_TEST_ENV) {
360
+ const testDir = fs.mkdtempSync(path.join(os.tmpdir(), "qwen_test_state_"));
361
+ process.env.QWEN_STATE_DIR = testDir;
362
+ return testDir;
363
+ }
364
+ if (IS_WINDOWS) {
365
+ const newDir = path.join(os.homedir(), ".castor");
366
+ migrateStateDir(path.join(os.homedir(), ".qwen"), newDir);
367
+ return newDir;
368
+ }
369
+ const winHomeWslPath = winHomeWsl();
370
+ if (winHomeWslPath) {
371
+ const winUserHomeCastor = path.join(winHomeWslPath, ".castor");
372
+ migrateStateDir(path.join(winHomeWslPath, ".qwen"), winUserHomeCastor);
373
+ try {
374
+ if (fs.existsSync(winUserHomeCastor)) return winUserHomeCastor;
375
+ } catch {}
376
+ }
377
+ const newDir = path.join(os.homedir(), ".castor");
378
+ migrateStateDir(path.join(os.homedir(), ".qwen"), newDir);
379
+ return newDir;
380
+ })();
381
+
382
+ export const TASK_DIR = path.join(QWEN_STATE_DIR, "tasks");
383
+ export const SLOTS_DIR = path.join(TASK_DIR, "slots");
384
+
385
+ // Global Configuration (~/.castor/config.json and ~/.castor/.env)
386
+ const GLOBAL_CONFIG_FILE = path.join(QWEN_STATE_DIR, "config.json");
387
+ const GLOBAL_ENV_FILE = path.join(QWEN_STATE_DIR, ".env");
388
+
389
+ /**
390
+ * Loads the machine-wide global configuration from ~/.castor/config.json or
391
+ * ~/.castor/.env. Single source of truth across all MCP host sessions
392
+ * (Claude Code, Antigravity, Cursor).
393
+ *
394
+ * Recognized top-level keys:
395
+ * - model: model name served by the engine (env: QWEN_MODEL)
396
+ * - baseURL: OpenAI-compatible endpoint (env: QWEN_BASE_URL)
397
+ * - max_context: nominal context window in tokens (env: QWEN_MAX_CONTEXT)
398
+ * - launch_command: shell command used to (re)start the engine (env: QWEN_LAUNCH_COMMAND)
399
+ * - tool_prefix: prefix applied to registered MCP tool names (env: MCP_TOOL_PREFIX)
400
+ * - api_key: engine API key (env: QWEN_API_KEY / OPENAI_API_KEY)
401
+ * - engine_type: engine type tag (env: QWEN_ENGINE_TYPE)
402
+ * - stop_command: shell command used to stop the engine (env: QWEN_STOP_COMMAND)
403
+ * - engine_log_path: path to the engine log (env: QWEN_LOG_PATH)
404
+ * - use_stream_proxy: enable/disable the stream proxy (env: USE_STREAM_PROXY)
405
+ * - stream_proxy_port: stream proxy port (env: STREAM_PROXY_PORT / VLLM_PROXY_PORT)
406
+ * - status_port: status server port (env: STATUS_PORT)
407
+ * - vllm_port: vLLM engine port (env: VLLM_PORT)
408
+ * - search: { provider, brave_api_key, tavily_api_key, context7_api_key, searxng_url }
409
+ *
410
+ * @returns {{
411
+ * model?: string,
412
+ * baseURL?: string,
413
+ * max_context?: number,
414
+ * launch_command?: string,
415
+ * tool_prefix?: string,
416
+ * api_key?: string,
417
+ * engine_type?: string,
418
+ * stop_command?: string,
419
+ * engine_log_path?: string,
420
+ * use_stream_proxy?: boolean,
421
+ * stream_proxy_port?: number,
422
+ * status_port?: number,
423
+ * vllm_port?: number,
424
+ * search: { provider?: string, brave_api_key?: string, tavily_api_key?: string, context7_api_key?: string, searxng_url?: string }
425
+ * }}
426
+ */
427
+ export function loadGlobalConfig() {
428
+ const config = { search: {} };
429
+ try {
430
+ if (fs.existsSync(GLOBAL_CONFIG_FILE)) {
431
+ const raw = fs.readFileSync(GLOBAL_CONFIG_FILE, "utf8").replace(/^\uFEFF/, "");
432
+ const parsed = JSON.parse(raw);
433
+ if (parsed && typeof parsed === "object") {
434
+ if (typeof parsed.model === "string" && parsed.model) config.model = parsed.model;
435
+ if (typeof parsed.baseURL === "string" && parsed.baseURL) config.baseURL = parsed.baseURL;
436
+ if (typeof parsed.max_context === "number" && parsed.max_context > 0) {
437
+ config.max_context = parsed.max_context;
438
+ }
439
+ if (typeof parsed.launch_command === "string" && parsed.launch_command) {
440
+ config.launch_command = parsed.launch_command;
441
+ }
442
+ if (typeof parsed.tool_prefix === "string" && parsed.tool_prefix) {
443
+ config.tool_prefix = parsed.tool_prefix;
444
+ }
445
+ if (typeof parsed.api_key === "string" && parsed.api_key) {
446
+ config.api_key = parsed.api_key;
447
+ }
448
+ if (typeof parsed.engine_type === "string" && parsed.engine_type) {
449
+ config.engine_type = parsed.engine_type;
450
+ }
451
+ if (typeof parsed.stop_command === "string" && parsed.stop_command) {
452
+ config.stop_command = parsed.stop_command;
453
+ }
454
+ if (typeof parsed.engine_log_path === "string" && parsed.engine_log_path) {
455
+ config.engine_log_path = parsed.engine_log_path;
456
+ }
457
+ if (typeof parsed.use_stream_proxy === "boolean") {
458
+ config.use_stream_proxy = parsed.use_stream_proxy;
459
+ }
460
+ if (typeof parsed.stream_proxy_port === "number" && parsed.stream_proxy_port > 0) {
461
+ config.stream_proxy_port = parsed.stream_proxy_port;
462
+ }
463
+ if (typeof parsed.status_port === "number" && parsed.status_port > 0) {
464
+ config.status_port = parsed.status_port;
465
+ }
466
+ if (typeof parsed.vllm_port === "number" && parsed.vllm_port > 0) {
467
+ config.vllm_port = parsed.vllm_port;
468
+ }
469
+ if (parsed.search && typeof parsed.search === "object") {
470
+ Object.assign(config.search, parsed.search);
471
+ }
472
+ }
473
+ }
474
+ } catch (err) {
475
+ process.stderr.write(`[config] Corrupt ${GLOBAL_CONFIG_FILE}: ${err.message}; defaults apply.\n`);
476
+ }
477
+
478
+ try {
479
+ if (fs.existsSync(GLOBAL_ENV_FILE)) {
480
+ const raw = fs.readFileSync(GLOBAL_ENV_FILE, "utf8").replace(/^\uFEFF/, "");
481
+ const lines = raw.split("\n");
482
+ for (const line of lines) {
483
+ const match = line.match(/^\s*([A-Za-z0-9_]+)\s*=\s*["']?([^"']+)["']?\s*(#.*)?$/);
484
+ if (match) {
485
+ const k = match[1];
486
+ const v = match[2].trim();
487
+ if (k === "BRAVE_API_KEY") config.search.brave_api_key = config.search.brave_api_key || v;
488
+ else if (k === "TAVILY_API_KEY") config.search.tavily_api_key = config.search.tavily_api_key || v;
489
+ else if (k === "CONTEXT7_API_KEY") config.search.context7_api_key = config.search.context7_api_key || v;
490
+ else if (k === "SEARXNG_URL") config.search.searxng_url = config.search.searxng_url || v;
491
+ else if (k === "SEARCH_PROVIDER") config.search.provider = config.search.provider || v;
492
+ else if (k === "QWEN_API_KEY") config.api_key = config.api_key || v;
493
+ else if (k === "QWEN_ENGINE_TYPE") config.engine_type = config.engine_type || v;
494
+ else if (k === "QWEN_STOP_COMMAND") config.stop_command = config.stop_command || v;
495
+ else if (k === "QWEN_LOG_PATH") config.engine_log_path = config.engine_log_path || v;
496
+ else if (k === "USE_STREAM_PROXY") config.use_stream_proxy = config.use_stream_proxy ?? (v !== "false" && v !== "0");
497
+ else if (k === "STREAM_PROXY_PORT" || k === "VLLM_PROXY_PORT") {
498
+ const p = parseInt(v, 10);
499
+ if (Number.isFinite(p) && p > 0) config.stream_proxy_port = config.stream_proxy_port ?? p;
500
+ } else if (k === "STATUS_PORT") {
501
+ const p = parseInt(v, 10);
502
+ if (Number.isFinite(p) && p > 0) config.status_port = config.status_port ?? p;
503
+ } else if (k === "VLLM_PORT") {
504
+ const p = parseInt(v, 10);
505
+ if (Number.isFinite(p) && p > 0) config.vllm_port = config.vllm_port ?? p;
506
+ }
507
+ }
508
+ }
509
+ }
510
+ } catch (err) {
511
+ process.stderr.write(`[config] Corrupt ${GLOBAL_ENV_FILE}: ${err.message}; defaults apply.\n`);
512
+ }
513
+
514
+ return config;
515
+ }
516
+
517
+ /**
518
+ * Applies global configuration credentials to process.env if not already present.
519
+ */
520
+ export function applyGlobalConfigToEnv() {
521
+ const cfg = loadGlobalConfig();
522
+ if (cfg.search?.brave_api_key && !process.env.BRAVE_API_KEY) {
523
+ process.env.BRAVE_API_KEY = cfg.search.brave_api_key;
524
+ }
525
+ if (cfg.search?.tavily_api_key && !process.env.TAVILY_API_KEY) {
526
+ process.env.TAVILY_API_KEY = cfg.search.tavily_api_key;
527
+ }
528
+ if (cfg.search?.context7_api_key && !process.env.CONTEXT7_API_KEY) {
529
+ process.env.CONTEXT7_API_KEY = cfg.search.context7_api_key;
530
+ }
531
+ if (cfg.search?.searxng_url && !process.env.SEARXNG_URL) {
532
+ process.env.SEARXNG_URL = cfg.search.searxng_url;
533
+ }
534
+ if (cfg.search?.provider && !process.env.SEARCH_PROVIDER) {
535
+ process.env.SEARCH_PROVIDER = cfg.search.provider;
536
+ }
537
+ }
538
+ applyGlobalConfigToEnv();
539
+
540
+ export function getSearchConfig() {
541
+ const cfg = loadGlobalConfig();
542
+ return {
543
+ provider: process.env.SEARCH_PROVIDER || cfg.search?.provider || "auto",
544
+ brave_api_key: process.env.BRAVE_API_KEY || cfg.search?.brave_api_key || "",
545
+ tavily_api_key: process.env.TAVILY_API_KEY || cfg.search?.tavily_api_key || "",
546
+ context7_api_key: process.env.CONTEXT7_API_KEY || cfg.search?.context7_api_key || "",
547
+ searxng_url: process.env.SEARXNG_URL || cfg.search?.searxng_url || "",
548
+ };
549
+ }
550
+
551
+ const _initSearchConfig = getSearchConfig();
552
+ const SEARCH_PROVIDER = _initSearchConfig.provider;
553
+ export const BRAVE_API_KEY = _initSearchConfig.brave_api_key;
554
+ export const TAVILY_API_KEY = _initSearchConfig.tavily_api_key;
555
+ const CONTEXT7_API_KEY = _initSearchConfig.context7_api_key;
556
+ const SEARXNG_URL = _initSearchConfig.searxng_url;
557
+
558
+ /**
559
+ * Maximum characters returned by web_fetch before boundary-aware truncation.
560
+ * - Unit: characters
561
+ * - Default: 60000
562
+ * - Override: QWEN_MAX_FETCH_CHARS
563
+ * @type {number}
564
+ */
565
+ export const MAX_FETCH_CHARS = (() => {
566
+ const parsed = parseInt(process.env.QWEN_MAX_FETCH_CHARS, 10);
567
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : 60_000;
568
+ })();
569
+
570
+
571
+ /**
572
+ * Inactivity threshold for engine telemetry before declaring an engine wedge.
573
+ * - Unit: seconds
574
+ * - Default: 120
575
+ * - Override: QWEN_WEDGE_SILENCE_S
576
+ * @type {number}
577
+ */
578
+ export const WEDGE_STATS_SILENCE_S = process.env.QWEN_WEDGE_SILENCE_S
579
+ ? parseInt(process.env.QWEN_WEDGE_SILENCE_S, 10)
580
+ : 120;
581
+
582
+ /**
583
+ * Flag enabling automated restart and recovery of a wedged inference engine.
584
+ * - Default: true
585
+ * - Override: QWEN_AUTO_HEAL ("0" disables)
586
+ * @type {boolean}
587
+ */
588
+ export const AUTO_HEAL = process.env.QWEN_AUTO_HEAL !== "0";
589
+
590
+ /**
591
+ * Lockfile path for engine auto-heal serialization.
592
+ * @type {string}
593
+ */
594
+ export const HEAL_LOCK_FILE = path.join(TASK_DIR, ".engine_heal.lock");
595
+
596
+ /**
597
+ * Time-to-live for engine heal lock acquisition.
598
+ * - Unit: milliseconds
599
+ * - Default: 300000 (5 minutes)
600
+ * @type {number}
601
+ */
602
+ export const HEAL_LOCK_TTL_MS = 5 * 60_000;
603
+
604
+ /**
605
+ * Lockfile path for exclusive engine boot coordination across processes.
606
+ * @type {string}
607
+ */
608
+ export const ENGINE_BOOT_LOCK_FILE = path.join(TASK_DIR, ".engine_boot.lock");
609
+
610
+ /**
611
+ * Time-to-live for engine boot lock acquisition.
612
+ * - Unit: milliseconds
613
+ * - Default: 480000 (8 minutes)
614
+ * @type {number}
615
+ */
616
+ export const ENGINE_BOOT_LOCK_TTL_MS = BOOT_TIMEOUT_MS;
617
+
618
+ /**
619
+ * Destination path for background engine startup logs.
620
+ * - Default: "/tmp/mcp_launch_huge.log"
621
+ * - Override: QWEN_LOG_PATH
622
+ * @type {string}
623
+ */
624
+ export const ENGINE_LOG_PATH = (() => {
625
+ const cfg = loadGlobalConfig();
626
+ return process.env.QWEN_LOG_PATH || cfg.engine_log_path || "/tmp/mcp_launch_huge.log";
627
+ })();
628
+
629
+ /**
630
+ * Counter ledger tracking cumulative engine wedge and restart events.
631
+ * @type {string}
632
+ */
633
+ export const WEDGE_COUNTER_FILE = path.join(TASK_DIR, ".wedge_counter.json");
634
+
635
+ /**
636
+ * Maximum allowable payload size accepted by the stream proxy.
637
+ * - Unit: bytes
638
+ * - Value: 52428800 (50 MB)
639
+ * @type {number}
640
+ */
641
+ export const PROXY_MAX_BODY_BYTES = 50 * 1024 * 1024;
642
+
643
+ /**
644
+ * Guard flag permitting tests or management commands to interrupt, probe, or reboot
645
+ * the live vLLM inference engine. Disabled by default to protect running workloads.
646
+ * - Default: false
647
+ * - Override: ALLOW_ENGINE_INTERRUPT ("1" enables)
648
+ * @type {boolean}
649
+ */
650
+ export const ALLOW_ENGINE_INTERRUPT = process.env.ALLOW_ENGINE_INTERRUPT === "1";
651
+
652
+ /**
653
+ * Base turn ceiling before requiring supervisor lease extension or cooperative landing.
654
+ * - Unit: count
655
+ * - Default: 80
656
+ * - Override: QWEN_BASE_TURN_BUDGET
657
+ * @type {number}
658
+ */
659
+ const DEFAULT_BASE_TURN_BUDGET = 80;
660
+ export const BASE_TURN_BUDGET = (() => {
661
+ const parsed = parseInt(process.env.QWEN_BASE_TURN_BUDGET, 10);
662
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_BASE_TURN_BUDGET;
663
+ })();
664
+
665
+ /**
666
+ * Maximum elastic turn ceiling reachable through supervisor lease extensions.
667
+ * - Unit: count
668
+ * - Default: 200
669
+ * - Override: QWEN_MAX_ELASTIC_TURNS
670
+ * @type {number}
671
+ */
672
+ const DEFAULT_MAX_ELASTIC_TURNS = 200;
673
+ export const MAX_ELASTIC_TURNS = (() => {
674
+ const parsed = parseInt(process.env.QWEN_MAX_ELASTIC_TURNS, 10);
675
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_MAX_ELASTIC_TURNS;
676
+ })();
677
+
678
+ /**
679
+ * Maximum allowable GPU KV cache utilization percentage before disallowing lease extension.
680
+ * - Unit: percentage (0-100)
681
+ * - Default: 85.0
682
+ * - Override: QWEN_KV_CACHE_HEADROOM_CEILING
683
+ * @type {number}
684
+ */
685
+ const DEFAULT_KV_CACHE_HEADROOM_CEILING = 85.0;
686
+ export const KV_CACHE_HEADROOM_CEILING = (() => {
687
+ const parsed = parseFloat(process.env.QWEN_KV_CACHE_HEADROOM_CEILING);
688
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_KV_CACHE_HEADROOM_CEILING;
689
+ })();
690
+
691
+ /**
692
+ * Minimum average speculative decoding acceptance length required for lease extension.
693
+ * - Unit: tokens per draft step
694
+ * - Default: 2.5
695
+ * - Override: QWEN_SPEC_ACCEPTANCE_FLOOR
696
+ * @type {number}
697
+ */
698
+ const DEFAULT_SPEC_ACCEPTANCE_FLOOR = 2.5;
699
+ export const SPEC_ACCEPTANCE_FLOOR = (() => {
700
+ const parsed = parseFloat(process.env.QWEN_SPEC_ACCEPTANCE_FLOOR);
701
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_SPEC_ACCEPTANCE_FLOOR;
702
+ })();
703
+
704
+ /**
705
+ * Sliding window size for action-hash stagnation and loop detection.
706
+ * - Unit: count
707
+ * - Default: 6
708
+ * - Override: QWEN_LOOP_DETECTION_WINDOW
709
+ * @type {number}
710
+ */
711
+ export const DEFAULT_LOOP_DETECTION_WINDOW = 6;
712
+ export const LOOP_DETECTION_WINDOW = (() => {
713
+ const parsed = parseInt(process.env.QWEN_LOOP_DETECTION_WINDOW, 10);
714
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_LOOP_DETECTION_WINDOW;
715
+ })();
716
+
717
+ /**
718
+ * Threshold of repeated identical non-mutating actions within the detection window
719
+ * required to trigger loop detection circuit-breaking.
720
+ * - Unit: count
721
+ * - Default: 3
722
+ * - Override: QWEN_LOOP_DETECTION_REPETITIONS
723
+ * @type {number}
724
+ */
725
+ export const DEFAULT_LOOP_DETECTION_REPETITIONS = 3;
726
+ export const LOOP_DETECTION_REPETITIONS = (() => {
727
+ const parsed = parseInt(process.env.QWEN_LOOP_DETECTION_REPETITIONS, 10);
728
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_LOOP_DETECTION_REPETITIONS;
729
+ })();
730
+
731
+ /**
732
+ * Character length of recent activity summaries returned in status and wait telemetry endpoints.
733
+ * - Unit: characters
734
+ * - Default: 300
735
+ * - Override: QWEN_SUPERVISOR_PREVIEW_CHARS
736
+ * @type {number}
737
+ */
738
+ const DEFAULT_SUPERVISOR_PREVIEW_CHARS = 300;
739
+ export const SUPERVISOR_PREVIEW_CHARS = (() => {
740
+ const parsed = parseInt(process.env.QWEN_SUPERVISOR_PREVIEW_CHARS, 10);
741
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_SUPERVISOR_PREVIEW_CHARS;
742
+ })();
743
+
744
+ /**
745
+ * Optional hard upper bound on session turn count; null allows unbounded orchestrator steering.
746
+ * - Unit: count
747
+ * - Default: null
748
+ * - Override: QWEN_MAX_TURNS
749
+ * @type {number|null}
750
+ */
751
+ export const MAX_TURNS = process.env.QWEN_MAX_TURNS
752
+ ? parseInt(process.env.QWEN_MAX_TURNS, 10)
753
+ : null;
754
+
755
+ /**
756
+ * Maximum re-prompt attempts following a token-ceiling (finish_reason: "length") cutoff.
757
+ * - Unit: count
758
+ * - Default: 8
759
+ * - Override: QWEN_MAX_CONTINUATION_TURNS
760
+ * @type {number}
761
+ */
762
+ const DEFAULT_MAX_CONTINUATION_TURNS = 8;
763
+ export const MAX_CONTINUATION_TURNS = (() => {
764
+ const parsed = parseInt(process.env.QWEN_MAX_CONTINUATION_TURNS, 10);
765
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_MAX_CONTINUATION_TURNS;
766
+ })();
767
+
768
+ /**
769
+ * Maximum retry attempts when the inference engine returns an empty generation stream.
770
+ * - Unit: count
771
+ * - Default: 2
772
+ * - Override: QWEN_EMPTY_STREAM_RETRIES
773
+ * @type {number}
774
+ */
775
+ const DEFAULT_EMPTY_STREAM_RETRIES = 2;
776
+ export const EMPTY_STREAM_RETRIES = (() => {
777
+ const parsed = parseInt(process.env.QWEN_EMPTY_STREAM_RETRIES, 10);
778
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_EMPTY_STREAM_RETRIES;
779
+ })();
780
+
781
+ /**
782
+ * Retry attempts for empty generation streams when prompt characters exceed EMPTY_STREAM_RETRY_DEPTH_CHARS.
783
+ * - Unit: count
784
+ * - Default: 4
785
+ * - Override: QWEN_EMPTY_STREAM_RETRIES_DEEP
786
+ * @type {number}
787
+ */
788
+ const DEFAULT_EMPTY_STREAM_RETRIES_DEEP = 4;
789
+ export const EMPTY_STREAM_RETRIES_DEEP = (() => {
790
+ const parsed = parseInt(process.env.QWEN_EMPTY_STREAM_RETRIES_DEEP, 10);
791
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_EMPTY_STREAM_RETRIES_DEEP;
792
+ })();
793
+
794
+ /**
795
+ * Prompt character threshold that activates EMPTY_STREAM_RETRIES_DEEP.
796
+ * - Unit: characters
797
+ * - Default: 525000 (~150k tokens)
798
+ * - Override: QWEN_EMPTY_STREAM_RETRY_DEPTH_CHARS
799
+ * @type {number}
800
+ */
801
+ const DEFAULT_EMPTY_STREAM_RETRY_DEPTH_CHARS = 525_000;
802
+ export const EMPTY_STREAM_RETRY_DEPTH_CHARS = (() => {
803
+ const parsed = parseInt(process.env.QWEN_EMPTY_STREAM_RETRY_DEPTH_CHARS, 10);
804
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_EMPTY_STREAM_RETRY_DEPTH_CHARS;
805
+ })();
806
+
807
+ /**
808
+ * Base delay for exponential backoff between empty-stream retries.
809
+ * - Unit: milliseconds
810
+ * - Default: 2000
811
+ * - Override: QWEN_EMPTY_STREAM_RETRY_BACKOFF_BASE_MS
812
+ * @type {number}
813
+ */
814
+ const DEFAULT_EMPTY_STREAM_RETRY_BACKOFF_BASE_MS = 2000;
815
+ export const EMPTY_STREAM_RETRY_BACKOFF_BASE_MS = (() => {
816
+ const parsed = parseInt(process.env.QWEN_EMPTY_STREAM_RETRY_BACKOFF_BASE_MS, 10);
817
+ return Number.isFinite(parsed) && parsed >= 0 ? parsed : DEFAULT_EMPTY_STREAM_RETRY_BACKOFF_BASE_MS;
818
+ })();
819
+
820
+ const DEFAULT_EMPTY_STREAM_RETRY_BACKOFF_CAP_MS = 30_000;
821
+ export const EMPTY_STREAM_RETRY_BACKOFF_CAP_MS = (() => {
822
+ const parsed = parseInt(process.env.QWEN_EMPTY_STREAM_RETRY_BACKOFF_CAP_MS, 10);
823
+ return Number.isFinite(parsed) && parsed >= 0 ? parsed : DEFAULT_EMPTY_STREAM_RETRY_BACKOFF_CAP_MS;
824
+ })();
825
+
826
+ /**
827
+ * Task prompt character threshold that emits advisory prompt_over_budget telemetry.
828
+ * - Unit: characters
829
+ * - Default: 1500
830
+ * - Override: QWEN_PROMPT_BUDGET_CHARS
831
+ * @type {number}
832
+ */
833
+ export const DEFAULT_PROMPT_BUDGET_CHARS = 1500;
834
+ export const PROMPT_BUDGET_CHARS = (() => {
835
+ const parsed = parseInt(process.env.QWEN_PROMPT_BUDGET_CHARS, 10);
836
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_PROMPT_BUDGET_CHARS;
837
+ })();
838
+
839
+ /**
840
+ * Substantive text length required after stripping guard markers to avoid classification as degenerate.
841
+ * - Unit: characters
842
+ * - Default: 200
843
+ * - Override: QWEN_DEGENERATE_FINAL_SUBSTANTIVE_CHARS
844
+ * @type {number}
845
+ */
846
+ const DEFAULT_DEGENERATE_FINAL_SUBSTANTIVE_CHARS = 200;
847
+ export const DEGENERATE_FINAL_SUBSTANTIVE_CHARS = (() => {
848
+ const parsed = parseInt(process.env.QWEN_DEGENERATE_FINAL_SUBSTANTIVE_CHARS, 10);
849
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_DEGENERATE_FINAL_SUBSTANTIVE_CHARS;
850
+ })();
851
+
852
+ /**
853
+ * Maximum turn count within which a truncated final response is evaluated for degeneracy.
854
+ * - Unit: count
855
+ * - Default: 3
856
+ * - Override: QWEN_DEGENERATE_FINAL_MAX_TURNS
857
+ * @type {number}
858
+ */
859
+ const DEFAULT_DEGENERATE_FINAL_MAX_TURNS = 3;
860
+ export const DEGENERATE_FINAL_MAX_TURNS = (() => {
861
+ const parsed = parseInt(process.env.QWEN_DEGENERATE_FINAL_MAX_TURNS, 10);
862
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_DEGENERATE_FINAL_MAX_TURNS;
863
+ })();
864
+
865
+ /**
866
+ * Consecutive non-mutating shell execution threshold before emitting an advisory warning.
867
+ * - Unit: count
868
+ * - Default: 4
869
+ * - Override: QWEN_PROBE_BUDGET
870
+ * @type {number}
871
+ */
872
+ const DEFAULT_PROBE_BUDGET = 4;
873
+ export const PROBE_BUDGET = (() => {
874
+ const parsed = parseInt(process.env.QWEN_PROBE_BUDGET, 10);
875
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_PROBE_BUDGET;
876
+ })();
877
+
878
+ /**
879
+ * Cumulative turn threshold for emitting an advisory session_warning event.
880
+ * - Unit: count
881
+ * - Default: 60
882
+ * - Override: QWEN_SESSION_WARN_TURNS
883
+ * @type {number}
884
+ */
885
+ const DEFAULT_SESSION_TURNS_WARN = 60;
886
+ export const SESSION_TURNS_WARN = (() => {
887
+ const parsed = parseInt(process.env.QWEN_SESSION_WARN_TURNS, 10);
888
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_SESSION_TURNS_WARN;
889
+ })();
890
+
891
+ /**
892
+ * Cumulative turn threshold for emitting a session_turn_limit_recommended advisory event.
893
+ * - Unit: count
894
+ * - Default: 80
895
+ * - Override: QWEN_SESSION_RECOMMEND_TURNS
896
+ * @type {number}
897
+ */
898
+ const DEFAULT_SESSION_TURNS_RECOMMEND = 80;
899
+ export const SESSION_TURNS_RECOMMEND = (() => {
900
+ const parsed = parseInt(process.env.QWEN_SESSION_RECOMMEND_TURNS, 10);
901
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_SESSION_TURNS_RECOMMEND;
902
+ })();
903
+
904
+ /**
905
+ * Re-prefill prompt token threshold for emitting a context_depth_warning event.
906
+ * - Unit: tokens
907
+ * - Default: 65536
908
+ * - Override: QWEN_CONTEXT_WARN_TOKENS
909
+ * @type {number}
910
+ */
911
+ const DEFAULT_CONTEXT_WARN_TOKENS = 65536;
912
+ export const CONTEXT_WARN_TOKENS = (() => {
913
+ const parsed = parseInt(process.env.QWEN_CONTEXT_WARN_TOKENS, 10);
914
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_CONTEXT_WARN_TOKENS;
915
+ })();
916
+
917
+ /**
918
+ * High-watermark context threshold for proactive session rollover recommendations.
919
+ * Emits an advisory recommendation when prompt tokens exceed this threshold.
920
+ * - Unit: tokens
921
+ * - Default: 180000
922
+ * - Override: QWEN_CONTEXT_HIGH_WATERMARK_TOKENS
923
+ * @type {number}
924
+ */
925
+ const DEFAULT_CONTEXT_HIGH_WATERMARK_TOKENS = 180_000;
926
+ export const CONTEXT_HIGH_WATERMARK_TOKENS = (() => {
927
+ const parsed = parseInt(process.env.QWEN_CONTEXT_HIGH_WATERMARK_TOKENS, 10);
928
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_CONTEXT_HIGH_WATERMARK_TOKENS;
929
+ })();
930
+
931
+ /**
932
+ * Hard emergency ceiling for context tokens before halting further accumulation.
933
+ * - Unit: tokens
934
+ * - Default: 215000
935
+ * - Override: QWEN_CONTEXT_EMERGENCY_CEILING_TOKENS
936
+ * @type {number}
937
+ */
938
+ const DEFAULT_CONTEXT_EMERGENCY_CEILING_TOKENS = 215_000;
939
+ export const CONTEXT_EMERGENCY_CEILING_TOKENS = (() => {
940
+ const parsed = parseInt(process.env.QWEN_CONTEXT_EMERGENCY_CEILING_TOKENS, 10);
941
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_CONTEXT_EMERGENCY_CEILING_TOKENS;
942
+ })();
943
+
944
+ /**
945
+ * Maximum file read size enforced when context high-watermark is active.
946
+ * - Unit: bytes
947
+ * - Default: 16384 (16 KB)
948
+ * - Override: QWEN_READ_GOVERNOR_MAX_BYTES
949
+ * @type {number}
950
+ */
951
+ const DEFAULT_READ_GOVERNOR_MAX_BYTES = 16 * 1024;
952
+ export const READ_GOVERNOR_MAX_BYTES = (() => {
953
+ const parsed = parseInt(process.env.QWEN_READ_GOVERNOR_MAX_BYTES, 10);
954
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_READ_GOVERNOR_MAX_BYTES;
955
+ })();
956
+
957
+ /**
958
+ * Tool result size threshold above which payload is spilled to disk.
959
+ * - Unit: bytes
960
+ * - Default: 16384 (16 KB)
961
+ * - Override: QWEN_TOOL_SPILL_BYTES
962
+ * @type {number}
963
+ */
964
+ const DEFAULT_TOOL_SPILL_BYTES = 16384;
965
+ export const TOOL_SPILL_BYTES = (() => {
966
+ const parsed = parseInt(process.env.QWEN_TOOL_SPILL_BYTES, 10);
967
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_TOOL_SPILL_BYTES;
968
+ })();
969
+
970
+ /**
971
+ * Generation token limit for bounded findings extraction on deliberation budget exhaustion.
972
+ * - Unit: tokens
973
+ * - Default: 4096
974
+ * - Override: QWEN_SALVAGE_MAX_TOKENS
975
+ * @type {number}
976
+ */
977
+ const DEFAULT_SALVAGE_MAX_TOKENS = 4096;
978
+ export const SALVAGE_MAX_TOKENS = (() => {
979
+ const parsed = parseInt(process.env.QWEN_SALVAGE_MAX_TOKENS, 10);
980
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_SALVAGE_MAX_TOKENS;
981
+ })();
982
+
983
+ // ---------------------------------------------------------------------------
984
+ // Engine configuration: model / baseURL / max_context / launch_command
985
+ // Precedence: environment variables > ~/.castor/config.json > built-in defaults.
986
+ // ---------------------------------------------------------------------------
987
+
988
+ const DEFAULT_MODEL = "Qwen3.8-27B";
989
+ const DEFAULT_MAX_CONTEXT = MAX_LEN_HUGE;
990
+ const DEFAULT_LAUNCH_COMMAND = "";
991
+
992
+ /**
993
+ * Resolves the engine configuration with precedence:
994
+ * 1. Environment variables (CASTOR_MODEL/QWEN_MODEL, CASTOR_BASE_URL/QWEN_BASE_URL,
995
+ * CASTOR_MAX_CONTEXT/QWEN_MAX_CONTEXT, CASTOR_LAUNCH_COMMAND/QWEN_LAUNCH_COMMAND,
996
+ * CASTOR_TOOL_PREFIX/MCP_TOOL_PREFIX)
997
+ * 2. Active profile in ~/.castor/config.json (profiles[active_profile])
998
+ * 3. ~/.castor/config.json root keys (model, baseURL, max_context, launch_command, tool_prefix)
999
+ * 4. Built-in defaults
1000
+ *
1001
+ * @returns {{
1002
+ * model: string,
1003
+ * baseURL: string,
1004
+ * max_context: number,
1005
+ * launch_command: string,
1006
+ * tool_prefix: string,
1007
+ * api_key?: string,
1008
+ * engine_type?: string,
1009
+ * stop_command?: string,
1010
+ * engine_log_path?: string
1011
+ * }}
1012
+ */
1013
+ export function getEngineConfig() {
1014
+ const cfg = loadGlobalConfig();
1015
+ const profile =
1016
+ (cfg.active_profile && cfg.profiles && cfg.profiles[cfg.active_profile]) ||
1017
+ {};
1018
+ const merged = { ...cfg, ...profile };
1019
+
1020
+ return {
1021
+ model:
1022
+ process.env.CASTOR_MODEL ||
1023
+ process.env.QWEN_MODEL ||
1024
+ merged.model ||
1025
+ DEFAULT_MODEL,
1026
+ baseURL:
1027
+ process.env.CASTOR_BASE_URL ||
1028
+ process.env.QWEN_BASE_URL ||
1029
+ merged.baseURL ||
1030
+ BASE_URL,
1031
+ max_context:
1032
+ parseInt(
1033
+ process.env.CASTOR_MAX_CONTEXT || process.env.QWEN_MAX_CONTEXT,
1034
+ 10
1035
+ ) ||
1036
+ merged.max_context ||
1037
+ DEFAULT_MAX_CONTEXT,
1038
+ launch_command:
1039
+ process.env.CASTOR_LAUNCH_COMMAND ||
1040
+ process.env.QWEN_LAUNCH_COMMAND ||
1041
+ merged.launch_command ||
1042
+ DEFAULT_LAUNCH_COMMAND,
1043
+ tool_prefix:
1044
+ process.env.CASTOR_TOOL_PREFIX ??
1045
+ process.env.MCP_TOOL_PREFIX ??
1046
+ merged.tool_prefix ??
1047
+ "castor",
1048
+ api_key:
1049
+ process.env.CASTOR_API_KEY ||
1050
+ process.env.QWEN_API_KEY ||
1051
+ merged.api_key ||
1052
+ "",
1053
+ engine_type:
1054
+ process.env.CASTOR_ENGINE_TYPE ||
1055
+ process.env.QWEN_ENGINE_TYPE ||
1056
+ merged.engine_type ||
1057
+ "vllm",
1058
+ stop_command:
1059
+ process.env.CASTOR_STOP_COMMAND ||
1060
+ process.env.QWEN_STOP_COMMAND ||
1061
+ merged.stop_command ||
1062
+ "",
1063
+ engine_log_path:
1064
+ process.env.CASTOR_LOG_PATH ||
1065
+ process.env.QWEN_LOG_PATH ||
1066
+ merged.engine_log_path ||
1067
+ "/tmp/mcp_launch_huge.log",
1068
+ };
1069
+ }
1070
+
1071
+ /**
1072
+ * Model name served by the local engine.
1073
+ * - Default: "Qwen3.8-27B"
1074
+ * - Override: CASTOR_MODEL / QWEN_MODEL (env) or `model` in ~/.castor/config.json
1075
+ * @type {string}
1076
+ */
1077
+ export const MODEL = getEngineConfig().model;
1078
+
1079
+ /**
1080
+ * Nominal maximum context window for the served model, in tokens.
1081
+ * - Default: 245760
1082
+ * - Override: CASTOR_MAX_CONTEXT / QWEN_MAX_CONTEXT (env) or `max_context` in ~/.castor/config.json
1083
+ * @type {number}
1084
+ */
1085
+ export const MAX_CONTEXT = getEngineConfig().max_context;
1086
+
1087
+ /**
1088
+ * Shell command used to (re)start the inference engine.
1089
+ * - Default: "" (no managed launch)
1090
+ * - Override: CASTOR_LAUNCH_COMMAND / QWEN_LAUNCH_COMMAND (env) or `launch_command` in ~/.castor/config.json
1091
+ * @type {string}
1092
+ */
1093
+ export const LAUNCH_COMMAND = getEngineConfig().launch_command;
1094
+
1095
+ /**
1096
+ * Prefix applied to registered MCP tool names (e.g. "castor" -> "castor_coworker").
1097
+ * - Default: "castor"
1098
+ * - Override: CASTOR_TOOL_PREFIX / MCP_TOOL_PREFIX (env) or `tool_prefix` in ~/.castor/config.json
1099
+ * @type {string}
1100
+ */
1101
+ export const TOOL_PREFIX = getEngineConfig().tool_prefix;
1102
+
1103
+ // ---------------------------------------------------------------------------
1104
+ // Config-file-aware resolved values.
1105
+ //
1106
+ // These are the "effective" values that honor ~/.castor/config.json in addition
1107
+ // to environment variables. They are defined at the bottom of the file (after
1108
+ // GLOBAL_CONFIG_FILE / GLOBAL_ENV_FILE are initialized) so that calling
1109
+ // loadGlobalConfig() here is safe (no TDZ).
1110
+ //
1111
+ // Precedence: environment variable > config.json > built-in default.
1112
+ // ---------------------------------------------------------------------------
1113
+
1114
+ /**
1115
+ * Effective stream proxy port (env > config.json > default 18022).
1116
+ * @type {number}
1117
+ */
1118
+ export const STREAM_PROXY_PORT_RESOLVED = (() => {
1119
+ const envPort = process.env.STREAM_PROXY_PORT || process.env.VLLM_PROXY_PORT;
1120
+ if (envPort) {
1121
+ const p = parseInt(envPort, 10);
1122
+ if (Number.isFinite(p) && p > 0) return p;
1123
+ }
1124
+ const cfg = loadGlobalConfig();
1125
+ if (typeof cfg.stream_proxy_port === "number" && cfg.stream_proxy_port > 0) {
1126
+ return cfg.stream_proxy_port;
1127
+ }
1128
+ return 18022;
1129
+ })();
1130
+
1131
+ /**
1132
+ * Effective use-stream-proxy flag (env > config.json > default true).
1133
+ * @type {boolean}
1134
+ */
1135
+ export const USE_STREAM_PROXY_RESOLVED = (() => {
1136
+ if (process.env.USE_STREAM_PROXY !== undefined) {
1137
+ return process.env.USE_STREAM_PROXY !== "false" && process.env.USE_STREAM_PROXY !== "0";
1138
+ }
1139
+ const cfg = loadGlobalConfig();
1140
+ if (typeof cfg.use_stream_proxy === "boolean") return cfg.use_stream_proxy;
1141
+ return true;
1142
+ })();
1143
+
1144
+ /**
1145
+ * Effective status port (env > config.json > default 18021).
1146
+ * @type {number}
1147
+ */
1148
+ export const STATUS_PORT_RESOLVED = (() => {
1149
+ const envPort = process.env.STATUS_PORT;
1150
+ if (envPort) {
1151
+ const p = parseInt(envPort, 10);
1152
+ if (Number.isFinite(p) && p > 0) return p;
1153
+ }
1154
+ const cfg = loadGlobalConfig();
1155
+ if (typeof cfg.status_port === "number" && cfg.status_port > 0) return cfg.status_port;
1156
+ return 18021;
1157
+ })();
1158
+
1159
+ /**
1160
+ * Effective vLLM port (env > config.json > default 18020).
1161
+ * @type {number}
1162
+ */
1163
+ export const VLLM_PORT_RESOLVED = (() => {
1164
+ const envPort = process.env.VLLM_PORT;
1165
+ if (envPort) {
1166
+ const p = parseInt(envPort, 10);
1167
+ if (Number.isFinite(p) && p > 0) return p;
1168
+ }
1169
+ const cfg = loadGlobalConfig();
1170
+ if (typeof cfg.vllm_port === "number" && cfg.vllm_port > 0) return cfg.vllm_port;
1171
+ return 18020;
1172
+ })();
1173
+
1174
+ /**
1175
+ * Engine API key (env > config.json > empty string).
1176
+ * @type {string}
1177
+ */
1178
+ export const API_KEY = (() => {
1179
+ const cfg = loadGlobalConfig();
1180
+ return process.env.QWEN_API_KEY || cfg.api_key || "";
1181
+ })();
1182
+
1183
+ /**
1184
+ * Engine type tag (env > config.json > "vllm").
1185
+ * @type {string}
1186
+ */
1187
+ export const ENGINE_TYPE = (() => {
1188
+ const cfg = loadGlobalConfig();
1189
+ return process.env.QWEN_ENGINE_TYPE || cfg.engine_type || "vllm";
1190
+ })();
1191
+
1192
+ /**
1193
+ * Shell command used to stop the inference engine (env > config.json > "").
1194
+ * @type {string}
1195
+ */
1196
+ export const STOP_COMMAND = (() => {
1197
+ const cfg = loadGlobalConfig();
1198
+ return process.env.QWEN_STOP_COMMAND || cfg.stop_command || "";
1199
+ })();
1200
+
1201
+
1202
+
1203
+
1204
+