mcp-castor 2026.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +487 -0
- package/bin/castor.js +706 -0
- package/index.js +206 -0
- package/package.json +97 -0
- package/skills/canary-test-staging/SKILL.md +24 -0
- package/skills/evo-mutation-rollback/SKILL.md +29 -0
- package/skills/hypothesis-generation/SKILL.md +26 -0
- package/skills/traceback-condensing/SKILL.md +26 -0
- package/src/castor_runner.js +469 -0
- package/src/config.js +1204 -0
- package/src/env.js +10 -0
- package/src/evo_engine.js +214 -0
- package/src/harness/core/events.js +75 -0
- package/src/harness/core/kernel.js +209 -0
- package/src/harness/evo/evaluator.js +156 -0
- package/src/harness/evo/evo_operator.js +550 -0
- package/src/harness/evo/lineage_dag.js +383 -0
- package/src/harness/evo/trace_repair.js +173 -0
- package/src/harness/evo/watchdog.js +72 -0
- package/src/harness/loop_detector.js +135 -0
- package/src/harness/runner.js +1216 -0
- package/src/harness/services/ast_service.js +1813 -0
- package/src/harness/services/event_logger.js +275 -0
- package/src/harness/services/mcp_bridge.js +408 -0
- package/src/harness/services/provider_vllm.js +728 -0
- package/src/harness/services/sandbox_fs.js +1238 -0
- package/src/harness/services/searxng_lifecycle.js +254 -0
- package/src/harness/services/shell_executor.js +264 -0
- package/src/harness/services/shell_validator.js +506 -0
- package/src/harness/services/web_service.js +828 -0
- package/src/platform.js +344 -0
- package/src/repetition_detector.js +139 -0
- package/src/semaphore.js +373 -0
- package/src/server_lifecycle.js +781 -0
- package/src/skills.js +400 -0
- package/src/state_pruner.js +392 -0
- package/src/task_registry.js +1357 -0
- package/src/telemetry.js +638 -0
- package/src/tools.js +997 -0
- package/src/wsl_bridge.js +629 -0
- package/src/wsl_env.js +171 -0
- package/stream_proxy.js +453 -0
package/src/config.js
ADDED
|
@@ -0,0 +1,1204 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import os from "node:os";
|
|
3
|
+
import path from "node:path";
|
|
4
|
+
import { winHomeWsl } from "./wsl_env.js";
|
|
5
|
+
import { IS_WINDOWS } from "./env.js";
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Platform detection flag for Windows environments.
|
|
9
|
+
* Re-exported from leaf env.js for backward compatibility.
|
|
10
|
+
* @type {boolean}
|
|
11
|
+
*/
|
|
12
|
+
export { IS_WINDOWS };
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Port for the vLLM OpenAI-compatible server.
|
|
16
|
+
* - Unit: port number
|
|
17
|
+
* - Default: 18020
|
|
18
|
+
* - Override: VLLM_PORT
|
|
19
|
+
* @type {number}
|
|
20
|
+
*/
|
|
21
|
+
export const VLLM_PORT = parseInt(process.env.VLLM_PORT || "18020", 10);
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Port for the zero-turn status and long-poll wait HTTP server.
|
|
25
|
+
* - Unit: port number
|
|
26
|
+
* - Default: 18021
|
|
27
|
+
* - Override: STATUS_PORT
|
|
28
|
+
* @type {number}
|
|
29
|
+
*/
|
|
30
|
+
export const STATUS_PORT = parseInt(process.env.STATUS_PORT || "18021", 10);
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Port for the local SSE stream sanitizer proxy.
|
|
34
|
+
* - Unit: port number
|
|
35
|
+
* - Default: 18022
|
|
36
|
+
* - Override: STREAM_PROXY_PORT (preferred) or VLLM_PROXY_PORT (legacy)
|
|
37
|
+
* - Config-file override: `stream_proxy_port` in ~/.castor/config.json
|
|
38
|
+
* (resolved in STREAM_PROXY_PORT_RESOLVED at the bottom of this file)
|
|
39
|
+
* @type {number}
|
|
40
|
+
*/
|
|
41
|
+
export const STREAM_PROXY_PORT = parseInt(
|
|
42
|
+
process.env.STREAM_PROXY_PORT || process.env.VLLM_PROXY_PORT || "18022",
|
|
43
|
+
10
|
|
44
|
+
);
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Whether to use the stream proxy.
|
|
48
|
+
* - Default: true
|
|
49
|
+
* - Override: USE_STREAM_PROXY ("false" or "0" disables)
|
|
50
|
+
* - Config-file override: `use_stream_proxy` in ~/.castor/config.json
|
|
51
|
+
* (resolved in USE_STREAM_PROXY_RESOLVED at the bottom of this file)
|
|
52
|
+
* @type {boolean}
|
|
53
|
+
*/
|
|
54
|
+
export const USE_STREAM_PROXY =
|
|
55
|
+
process.env.USE_STREAM_PROXY !== "false" && process.env.USE_STREAM_PROXY !== "0";
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Base URL for vLLM API completions.
|
|
59
|
+
* @type {string}
|
|
60
|
+
*/
|
|
61
|
+
export const BASE_URL = `http://localhost:${VLLM_PORT}/v1`;
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Nominal maximum context length for the Qwen model architecture.
|
|
65
|
+
* - Unit: tokens
|
|
66
|
+
* - Value: 245760
|
|
67
|
+
* @type {number}
|
|
68
|
+
*/
|
|
69
|
+
export const MAX_LEN_HUGE = 245760;
|
|
70
|
+
/**
|
|
71
|
+
* Maximum duration to wait for the vLLM engine to become healthy after boot.
|
|
72
|
+
* - Unit: milliseconds
|
|
73
|
+
* - Default: 480000 (8 minutes)
|
|
74
|
+
* @type {number}
|
|
75
|
+
*/
|
|
76
|
+
export const BOOT_TIMEOUT_MS = 480_000;
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Polling cadence when checking vLLM health during startup.
|
|
80
|
+
* - Unit: milliseconds
|
|
81
|
+
* - Default: 3000 (3 seconds)
|
|
82
|
+
* @type {number}
|
|
83
|
+
*/
|
|
84
|
+
export const BOOT_POLL_MS = 3000;
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Maximum completion tokens allowed for a single generation turn.
|
|
88
|
+
* - Unit: tokens
|
|
89
|
+
* - Default: 49152
|
|
90
|
+
* - Override: QWEN_MAX_TOKENS
|
|
91
|
+
* @type {number}
|
|
92
|
+
*/
|
|
93
|
+
const DEFAULT_MAX_TOKENS = 49152;
|
|
94
|
+
export const MAX_TOKENS = (() => {
|
|
95
|
+
const parsed = parseInt(process.env.QWEN_MAX_TOKENS, 10);
|
|
96
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_MAX_TOKENS;
|
|
97
|
+
})();
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Retrieves the current default reasoning effort for generation dispatches.
|
|
101
|
+
* - Override: QWEN_REASONING_EFFORT
|
|
102
|
+
* - Default: "medium"
|
|
103
|
+
* @returns {"xhigh"|"medium"|"low"}
|
|
104
|
+
*/
|
|
105
|
+
export function getReasoningEffort() {
|
|
106
|
+
const v = process.env.QWEN_REASONING_EFFORT;
|
|
107
|
+
return v ? v : "medium";
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Supported reasoning-effort tiers accepted by the engine's chat template.
|
|
112
|
+
* Canonical enum for tool schemas and dispatch validation.
|
|
113
|
+
* @type {readonly string[]}
|
|
114
|
+
*/
|
|
115
|
+
export const REASONING_EFFORT_TIERS = ["xhigh", "medium", "low"];
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Synchronous race window before yielding to asynchronous zero-turn HTTP wait.
|
|
119
|
+
* - Unit: milliseconds
|
|
120
|
+
* - Default: 15000 (15 seconds)
|
|
121
|
+
* - Override: QWEN_RACE_MS
|
|
122
|
+
* @type {number}
|
|
123
|
+
*/
|
|
124
|
+
const DEFAULT_RACE_MS = 15_000;
|
|
125
|
+
export const RACE_MS = process.env.QWEN_RACE_MS
|
|
126
|
+
? parseInt(process.env.QWEN_RACE_MS, 10)
|
|
127
|
+
: DEFAULT_RACE_MS;
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* Default maximum execution timeout for a Castor background task.
|
|
131
|
+
* - Unit: milliseconds
|
|
132
|
+
* - Default: 14400000 (4 hours)
|
|
133
|
+
* @type {number}
|
|
134
|
+
*/
|
|
135
|
+
export const DEFAULT_TIMEOUT_MS = 14_400_000;
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* Enforced lower bound for task timeout configuration.
|
|
139
|
+
* - Unit: milliseconds
|
|
140
|
+
* - Default: 600000 (10 minutes)
|
|
141
|
+
* - Override: QWEN_MIN_TIMEOUT_MS
|
|
142
|
+
* @type {number}
|
|
143
|
+
*/
|
|
144
|
+
const DEFAULT_MIN_TIMEOUT_MS = 600_000;
|
|
145
|
+
export const MIN_TIMEOUT_MS = (() => {
|
|
146
|
+
const parsed = parseInt(process.env.QWEN_MIN_TIMEOUT_MS, 10);
|
|
147
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_MIN_TIMEOUT_MS;
|
|
148
|
+
})();
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Inactivity watchdog timeout for stream stalls or unhandled process locks.
|
|
152
|
+
* - Unit: milliseconds
|
|
153
|
+
* - Default: 1800000 (30 minutes)
|
|
154
|
+
* - Override: QWEN_INACTIVITY_TIMEOUT_MS
|
|
155
|
+
* @type {number}
|
|
156
|
+
*/
|
|
157
|
+
export const INACTIVITY_TIMEOUT_MS = (() => {
|
|
158
|
+
const parsed = parseInt(process.env.QWEN_INACTIVITY_TIMEOUT_MS, 10);
|
|
159
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : 1_800_000;
|
|
160
|
+
})();
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Timeout allowed before the initial streaming token is emitted by the engine.
|
|
164
|
+
* - Unit: milliseconds
|
|
165
|
+
* - Default: 240000 (4 minutes)
|
|
166
|
+
* - Override: QWEN_FIRST_TOKEN_TIMEOUT_MS
|
|
167
|
+
* @type {number}
|
|
168
|
+
*/
|
|
169
|
+
export const FIRST_TOKEN_TIMEOUT_MS = (() => {
|
|
170
|
+
const parsed = parseInt(process.env.QWEN_FIRST_TOKEN_TIMEOUT_MS, 10);
|
|
171
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : 240_000;
|
|
172
|
+
})();
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* Maximum allowed idle duration between consecutive stream chunks before aborting.
|
|
176
|
+
* - Unit: milliseconds
|
|
177
|
+
* - Default: 1200000 (20 minutes)
|
|
178
|
+
* - Override: QWEN_STREAM_IDLE_TIMEOUT_MS
|
|
179
|
+
* @type {number}
|
|
180
|
+
*/
|
|
181
|
+
const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 1_200_000; // 20 min
|
|
182
|
+
export const STREAM_IDLE_TIMEOUT_MS = (() => {
|
|
183
|
+
const parsed = parseInt(process.env.QWEN_STREAM_IDLE_TIMEOUT_MS, 10);
|
|
184
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_STREAM_IDLE_TIMEOUT_MS;
|
|
185
|
+
})();
|
|
186
|
+
|
|
187
|
+
/**
|
|
188
|
+
* Streaming idle timeout applied when prompt tokens exceed STREAM_IDLE_DEPTH_TOKENS.
|
|
189
|
+
* - Unit: milliseconds
|
|
190
|
+
* - Default: 2400000 (40 minutes)
|
|
191
|
+
* - Override: QWEN_STREAM_IDLE_TIMEOUT_DEEP_MS
|
|
192
|
+
* @type {number}
|
|
193
|
+
*/
|
|
194
|
+
const DEFAULT_STREAM_IDLE_TIMEOUT_MS_DEEP = 2_400_000; // 40 min
|
|
195
|
+
export const STREAM_IDLE_TIMEOUT_MS_DEEP = (() => {
|
|
196
|
+
const parsed = parseInt(process.env.QWEN_STREAM_IDLE_TIMEOUT_DEEP_MS, 10);
|
|
197
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_STREAM_IDLE_TIMEOUT_MS_DEEP;
|
|
198
|
+
})();
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* Prompt token depth threshold that activates STREAM_IDLE_TIMEOUT_MS_DEEP.
|
|
202
|
+
* - Unit: tokens
|
|
203
|
+
* - Default: 35000
|
|
204
|
+
* - Override: QWEN_STREAM_IDLE_DEPTH_TOKENS
|
|
205
|
+
* @type {number}
|
|
206
|
+
*/
|
|
207
|
+
const DEFAULT_STREAM_IDLE_DEPTH_TOKENS = 35_000;
|
|
208
|
+
export const STREAM_IDLE_DEPTH_TOKENS = (() => {
|
|
209
|
+
const parsed = parseInt(process.env.QWEN_STREAM_IDLE_DEPTH_TOKENS, 10);
|
|
210
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_STREAM_IDLE_DEPTH_TOKENS;
|
|
211
|
+
})();
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* Maximum reasoning tokens allowed per generation turn before terminating thinking.
|
|
215
|
+
* - Unit: tokens
|
|
216
|
+
* - Default: 32768
|
|
217
|
+
* - Override: QWEN_MAX_REASONING_TOKENS
|
|
218
|
+
* @type {number}
|
|
219
|
+
*/
|
|
220
|
+
const DEFAULT_MAX_REASONING_TOKENS = 32_768;
|
|
221
|
+
export const MAX_REASONING_TOKENS = (() => {
|
|
222
|
+
const parsed = parseInt(process.env.QWEN_MAX_REASONING_TOKENS, 10);
|
|
223
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_MAX_REASONING_TOKENS;
|
|
224
|
+
})();
|
|
225
|
+
|
|
226
|
+
/**
|
|
227
|
+
* Additional execution time granted per supervisor lease extension.
|
|
228
|
+
* - Unit: milliseconds
|
|
229
|
+
* - Default: 600000 (10 minutes)
|
|
230
|
+
* @type {number}
|
|
231
|
+
*/
|
|
232
|
+
export const EXTENSION_BONUS_TIMEOUT_MS = 600_000;
|
|
233
|
+
|
|
234
|
+
/**
|
|
235
|
+
* Minimum allowable task retention period (ms). Enforces that task JSON files
|
|
236
|
+
* outlive the maximum legal execution timeout plus inactivity watchdog margin.
|
|
237
|
+
* - Unit: milliseconds
|
|
238
|
+
* - Default: 16200000 (4h 30m)
|
|
239
|
+
* @type {number}
|
|
240
|
+
*/
|
|
241
|
+
export const TASK_RETENTION_FLOOR_MS = DEFAULT_TIMEOUT_MS + 1_800_000;
|
|
242
|
+
|
|
243
|
+
/**
|
|
244
|
+
* Task artifact retention period on disk before automated pruning. Clamped to
|
|
245
|
+
* {@link TASK_RETENTION_FLOOR_MS} to prevent premature deletion of active tasks.
|
|
246
|
+
* - Unit: milliseconds
|
|
247
|
+
* - Default: 604800000 (7 days)
|
|
248
|
+
* - Override: QWEN_TASK_RETENTION_MS
|
|
249
|
+
* @type {number}
|
|
250
|
+
*/
|
|
251
|
+
const DEFAULT_TASK_RETENTION_MS = 604_800_000;
|
|
252
|
+
export const TASK_RETENTION_MS = (() => {
|
|
253
|
+
const parsed = parseInt(process.env.QWEN_TASK_RETENTION_MS, 10);
|
|
254
|
+
const requested =
|
|
255
|
+
Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_TASK_RETENTION_MS;
|
|
256
|
+
return Math.max(requested, TASK_RETENTION_FLOOR_MS);
|
|
257
|
+
})();
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* Heartbeat staleness threshold for reaping orphaned tasks whose owner process is dead.
|
|
261
|
+
* - Unit: milliseconds
|
|
262
|
+
* - Default: 30000 (30 seconds)
|
|
263
|
+
* - Override: QWEN_ORPHAN_REAP_STALE_MS
|
|
264
|
+
* @type {number}
|
|
265
|
+
*/
|
|
266
|
+
const DEFAULT_ORPHAN_REAP_STALE_MS = 30_000;
|
|
267
|
+
export const ORPHAN_REAP_STALE_MS = (() => {
|
|
268
|
+
const parsed = parseInt(process.env.QWEN_ORPHAN_REAP_STALE_MS, 10);
|
|
269
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_ORPHAN_REAP_STALE_MS;
|
|
270
|
+
})();
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* Maximum number of concurrent tasks executed in parallel.
|
|
274
|
+
* - Unit: count
|
|
275
|
+
* - Default: 1
|
|
276
|
+
* - Override: QWEN_MAX_CONCURRENT
|
|
277
|
+
* @type {number}
|
|
278
|
+
*/
|
|
279
|
+
export const MAX_CONCURRENT_TASKS = process.env.QWEN_MAX_CONCURRENT
|
|
280
|
+
? Math.max(1, parseInt(process.env.QWEN_MAX_CONCURRENT, 10))
|
|
281
|
+
: 1;
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Heartbeat refresh interval for active task slot leases.
|
|
285
|
+
* - Unit: milliseconds
|
|
286
|
+
* - Default: 15000 (15 seconds)
|
|
287
|
+
* @type {number}
|
|
288
|
+
*/
|
|
289
|
+
export const SLOT_HEARTBEAT_MS = 15_000;
|
|
290
|
+
|
|
291
|
+
/**
|
|
292
|
+
* Inactivity duration before an unrefreshed slot lease is declared wedged.
|
|
293
|
+
* - Unit: milliseconds
|
|
294
|
+
* - Default: 300000 (5 minutes)
|
|
295
|
+
* @type {number}
|
|
296
|
+
*/
|
|
297
|
+
export const SLOT_WEDGED_MS = 300_000;
|
|
298
|
+
|
|
299
|
+
/**
|
|
300
|
+
* Polling cadence when waiting to acquire an exclusive slot lease.
|
|
301
|
+
* - Unit: milliseconds
|
|
302
|
+
* - Default: 1000 (1 second)
|
|
303
|
+
* @type {number}
|
|
304
|
+
*/
|
|
305
|
+
export const SLOT_POLL_MS = 1_000;
|
|
306
|
+
|
|
307
|
+
/**
|
|
308
|
+
* Identifies whether execution is occurring within an automated test suite.
|
|
309
|
+
* @type {boolean}
|
|
310
|
+
*/
|
|
311
|
+
export const IS_TEST_ENV = Boolean(
|
|
312
|
+
process.env.NODE_ENV === "test" ||
|
|
313
|
+
process.env.TEST_OFFLINE === "1" ||
|
|
314
|
+
(process.env.npm_lifecycle_event && process.env.npm_lifecycle_event.includes("test")) ||
|
|
315
|
+
process.argv.some((arg) => typeof arg === "string" && (arg.endsWith(".test.js") || arg.includes(".test.") || arg === "--test"))
|
|
316
|
+
);
|
|
317
|
+
|
|
318
|
+
if (IS_TEST_ENV && process.env.NODE_ENV !== "test") {
|
|
319
|
+
process.env.NODE_ENV = "test";
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
/**
|
|
323
|
+
* One-time migration from the legacy ~/.qwen state directory to ~/.castor.
|
|
324
|
+
* If the legacy directory exists and the new one does not, it is renamed
|
|
325
|
+
* (or copied as a fallback) so existing task state, config.json, and
|
|
326
|
+
* session ledgers carry over. Never throws; failures are logged to stderr.
|
|
327
|
+
*
|
|
328
|
+
* @param {string} legacyDir - legacy state dir (e.g. ~/.qwen)
|
|
329
|
+
* @param {string} newDir - new state dir (e.g. ~/.castor)
|
|
330
|
+
*/
|
|
331
|
+
function migrateStateDir(legacyDir, newDir) {
|
|
332
|
+
try {
|
|
333
|
+
if (!fs.existsSync(legacyDir) || fs.existsSync(newDir)) return;
|
|
334
|
+
try {
|
|
335
|
+
fs.renameSync(legacyDir, newDir);
|
|
336
|
+
} catch {
|
|
337
|
+
// Cross-device or permission failure: fall back to a recursive copy.
|
|
338
|
+
fs.cpSync(legacyDir, newDir, { recursive: true });
|
|
339
|
+
}
|
|
340
|
+
process.stderr.write(
|
|
341
|
+
`[config] Migrated state directory ${legacyDir} -> ${newDir}\n`
|
|
342
|
+
);
|
|
343
|
+
} catch (err) {
|
|
344
|
+
process.stderr.write(
|
|
345
|
+
`[config] State dir migration failed (${err.message}); using ${newDir}\n`
|
|
346
|
+
);
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
/**
|
|
351
|
+
* Root directory for task state, slot leases, and session ledgers.
|
|
352
|
+
* In test environments, allocates an isolated temporary scratchpad.
|
|
353
|
+
* In production environments, resolves to ~/.castor (or WSL host equivalent),
|
|
354
|
+
* with a one-time migration from the legacy ~/.qwen directory.
|
|
355
|
+
* - Override: QWEN_STATE_DIR
|
|
356
|
+
* @type {string}
|
|
357
|
+
*/
|
|
358
|
+
export const QWEN_STATE_DIR = process.env.QWEN_STATE_DIR || (() => {
|
|
359
|
+
if (IS_TEST_ENV) {
|
|
360
|
+
const testDir = fs.mkdtempSync(path.join(os.tmpdir(), "qwen_test_state_"));
|
|
361
|
+
process.env.QWEN_STATE_DIR = testDir;
|
|
362
|
+
return testDir;
|
|
363
|
+
}
|
|
364
|
+
if (IS_WINDOWS) {
|
|
365
|
+
const newDir = path.join(os.homedir(), ".castor");
|
|
366
|
+
migrateStateDir(path.join(os.homedir(), ".qwen"), newDir);
|
|
367
|
+
return newDir;
|
|
368
|
+
}
|
|
369
|
+
const winHomeWslPath = winHomeWsl();
|
|
370
|
+
if (winHomeWslPath) {
|
|
371
|
+
const winUserHomeCastor = path.join(winHomeWslPath, ".castor");
|
|
372
|
+
migrateStateDir(path.join(winHomeWslPath, ".qwen"), winUserHomeCastor);
|
|
373
|
+
try {
|
|
374
|
+
if (fs.existsSync(winUserHomeCastor)) return winUserHomeCastor;
|
|
375
|
+
} catch {}
|
|
376
|
+
}
|
|
377
|
+
const newDir = path.join(os.homedir(), ".castor");
|
|
378
|
+
migrateStateDir(path.join(os.homedir(), ".qwen"), newDir);
|
|
379
|
+
return newDir;
|
|
380
|
+
})();
|
|
381
|
+
|
|
382
|
+
export const TASK_DIR = path.join(QWEN_STATE_DIR, "tasks");
|
|
383
|
+
export const SLOTS_DIR = path.join(TASK_DIR, "slots");
|
|
384
|
+
|
|
385
|
+
// Global Configuration (~/.castor/config.json and ~/.castor/.env)
|
|
386
|
+
const GLOBAL_CONFIG_FILE = path.join(QWEN_STATE_DIR, "config.json");
|
|
387
|
+
const GLOBAL_ENV_FILE = path.join(QWEN_STATE_DIR, ".env");
|
|
388
|
+
|
|
389
|
+
/**
|
|
390
|
+
* Loads the machine-wide global configuration from ~/.castor/config.json or
|
|
391
|
+
* ~/.castor/.env. Single source of truth across all MCP host sessions
|
|
392
|
+
* (Claude Code, Antigravity, Cursor).
|
|
393
|
+
*
|
|
394
|
+
* Recognized top-level keys:
|
|
395
|
+
* - model: model name served by the engine (env: QWEN_MODEL)
|
|
396
|
+
* - baseURL: OpenAI-compatible endpoint (env: QWEN_BASE_URL)
|
|
397
|
+
* - max_context: nominal context window in tokens (env: QWEN_MAX_CONTEXT)
|
|
398
|
+
* - launch_command: shell command used to (re)start the engine (env: QWEN_LAUNCH_COMMAND)
|
|
399
|
+
* - tool_prefix: prefix applied to registered MCP tool names (env: MCP_TOOL_PREFIX)
|
|
400
|
+
* - api_key: engine API key (env: QWEN_API_KEY / OPENAI_API_KEY)
|
|
401
|
+
* - engine_type: engine type tag (env: QWEN_ENGINE_TYPE)
|
|
402
|
+
* - stop_command: shell command used to stop the engine (env: QWEN_STOP_COMMAND)
|
|
403
|
+
* - engine_log_path: path to the engine log (env: QWEN_LOG_PATH)
|
|
404
|
+
* - use_stream_proxy: enable/disable the stream proxy (env: USE_STREAM_PROXY)
|
|
405
|
+
* - stream_proxy_port: stream proxy port (env: STREAM_PROXY_PORT / VLLM_PROXY_PORT)
|
|
406
|
+
* - status_port: status server port (env: STATUS_PORT)
|
|
407
|
+
* - vllm_port: vLLM engine port (env: VLLM_PORT)
|
|
408
|
+
* - search: { provider, brave_api_key, tavily_api_key, context7_api_key, searxng_url }
|
|
409
|
+
*
|
|
410
|
+
* @returns {{
|
|
411
|
+
* model?: string,
|
|
412
|
+
* baseURL?: string,
|
|
413
|
+
* max_context?: number,
|
|
414
|
+
* launch_command?: string,
|
|
415
|
+
* tool_prefix?: string,
|
|
416
|
+
* api_key?: string,
|
|
417
|
+
* engine_type?: string,
|
|
418
|
+
* stop_command?: string,
|
|
419
|
+
* engine_log_path?: string,
|
|
420
|
+
* use_stream_proxy?: boolean,
|
|
421
|
+
* stream_proxy_port?: number,
|
|
422
|
+
* status_port?: number,
|
|
423
|
+
* vllm_port?: number,
|
|
424
|
+
* search: { provider?: string, brave_api_key?: string, tavily_api_key?: string, context7_api_key?: string, searxng_url?: string }
|
|
425
|
+
* }}
|
|
426
|
+
*/
|
|
427
|
+
export function loadGlobalConfig() {
|
|
428
|
+
const config = { search: {} };
|
|
429
|
+
try {
|
|
430
|
+
if (fs.existsSync(GLOBAL_CONFIG_FILE)) {
|
|
431
|
+
const raw = fs.readFileSync(GLOBAL_CONFIG_FILE, "utf8").replace(/^\uFEFF/, "");
|
|
432
|
+
const parsed = JSON.parse(raw);
|
|
433
|
+
if (parsed && typeof parsed === "object") {
|
|
434
|
+
if (typeof parsed.model === "string" && parsed.model) config.model = parsed.model;
|
|
435
|
+
if (typeof parsed.baseURL === "string" && parsed.baseURL) config.baseURL = parsed.baseURL;
|
|
436
|
+
if (typeof parsed.max_context === "number" && parsed.max_context > 0) {
|
|
437
|
+
config.max_context = parsed.max_context;
|
|
438
|
+
}
|
|
439
|
+
if (typeof parsed.launch_command === "string" && parsed.launch_command) {
|
|
440
|
+
config.launch_command = parsed.launch_command;
|
|
441
|
+
}
|
|
442
|
+
if (typeof parsed.tool_prefix === "string" && parsed.tool_prefix) {
|
|
443
|
+
config.tool_prefix = parsed.tool_prefix;
|
|
444
|
+
}
|
|
445
|
+
if (typeof parsed.api_key === "string" && parsed.api_key) {
|
|
446
|
+
config.api_key = parsed.api_key;
|
|
447
|
+
}
|
|
448
|
+
if (typeof parsed.engine_type === "string" && parsed.engine_type) {
|
|
449
|
+
config.engine_type = parsed.engine_type;
|
|
450
|
+
}
|
|
451
|
+
if (typeof parsed.stop_command === "string" && parsed.stop_command) {
|
|
452
|
+
config.stop_command = parsed.stop_command;
|
|
453
|
+
}
|
|
454
|
+
if (typeof parsed.engine_log_path === "string" && parsed.engine_log_path) {
|
|
455
|
+
config.engine_log_path = parsed.engine_log_path;
|
|
456
|
+
}
|
|
457
|
+
if (typeof parsed.use_stream_proxy === "boolean") {
|
|
458
|
+
config.use_stream_proxy = parsed.use_stream_proxy;
|
|
459
|
+
}
|
|
460
|
+
if (typeof parsed.stream_proxy_port === "number" && parsed.stream_proxy_port > 0) {
|
|
461
|
+
config.stream_proxy_port = parsed.stream_proxy_port;
|
|
462
|
+
}
|
|
463
|
+
if (typeof parsed.status_port === "number" && parsed.status_port > 0) {
|
|
464
|
+
config.status_port = parsed.status_port;
|
|
465
|
+
}
|
|
466
|
+
if (typeof parsed.vllm_port === "number" && parsed.vllm_port > 0) {
|
|
467
|
+
config.vllm_port = parsed.vllm_port;
|
|
468
|
+
}
|
|
469
|
+
if (parsed.search && typeof parsed.search === "object") {
|
|
470
|
+
Object.assign(config.search, parsed.search);
|
|
471
|
+
}
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
} catch (err) {
|
|
475
|
+
process.stderr.write(`[config] Corrupt ${GLOBAL_CONFIG_FILE}: ${err.message}; defaults apply.\n`);
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
try {
|
|
479
|
+
if (fs.existsSync(GLOBAL_ENV_FILE)) {
|
|
480
|
+
const raw = fs.readFileSync(GLOBAL_ENV_FILE, "utf8").replace(/^\uFEFF/, "");
|
|
481
|
+
const lines = raw.split("\n");
|
|
482
|
+
for (const line of lines) {
|
|
483
|
+
const match = line.match(/^\s*([A-Za-z0-9_]+)\s*=\s*["']?([^"']+)["']?\s*(#.*)?$/);
|
|
484
|
+
if (match) {
|
|
485
|
+
const k = match[1];
|
|
486
|
+
const v = match[2].trim();
|
|
487
|
+
if (k === "BRAVE_API_KEY") config.search.brave_api_key = config.search.brave_api_key || v;
|
|
488
|
+
else if (k === "TAVILY_API_KEY") config.search.tavily_api_key = config.search.tavily_api_key || v;
|
|
489
|
+
else if (k === "CONTEXT7_API_KEY") config.search.context7_api_key = config.search.context7_api_key || v;
|
|
490
|
+
else if (k === "SEARXNG_URL") config.search.searxng_url = config.search.searxng_url || v;
|
|
491
|
+
else if (k === "SEARCH_PROVIDER") config.search.provider = config.search.provider || v;
|
|
492
|
+
else if (k === "QWEN_API_KEY") config.api_key = config.api_key || v;
|
|
493
|
+
else if (k === "QWEN_ENGINE_TYPE") config.engine_type = config.engine_type || v;
|
|
494
|
+
else if (k === "QWEN_STOP_COMMAND") config.stop_command = config.stop_command || v;
|
|
495
|
+
else if (k === "QWEN_LOG_PATH") config.engine_log_path = config.engine_log_path || v;
|
|
496
|
+
else if (k === "USE_STREAM_PROXY") config.use_stream_proxy = config.use_stream_proxy ?? (v !== "false" && v !== "0");
|
|
497
|
+
else if (k === "STREAM_PROXY_PORT" || k === "VLLM_PROXY_PORT") {
|
|
498
|
+
const p = parseInt(v, 10);
|
|
499
|
+
if (Number.isFinite(p) && p > 0) config.stream_proxy_port = config.stream_proxy_port ?? p;
|
|
500
|
+
} else if (k === "STATUS_PORT") {
|
|
501
|
+
const p = parseInt(v, 10);
|
|
502
|
+
if (Number.isFinite(p) && p > 0) config.status_port = config.status_port ?? p;
|
|
503
|
+
} else if (k === "VLLM_PORT") {
|
|
504
|
+
const p = parseInt(v, 10);
|
|
505
|
+
if (Number.isFinite(p) && p > 0) config.vllm_port = config.vllm_port ?? p;
|
|
506
|
+
}
|
|
507
|
+
}
|
|
508
|
+
}
|
|
509
|
+
}
|
|
510
|
+
} catch (err) {
|
|
511
|
+
process.stderr.write(`[config] Corrupt ${GLOBAL_ENV_FILE}: ${err.message}; defaults apply.\n`);
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
return config;
|
|
515
|
+
}
|
|
516
|
+
|
|
517
|
+
/**
|
|
518
|
+
* Applies global configuration credentials to process.env if not already present.
|
|
519
|
+
*/
|
|
520
|
+
export function applyGlobalConfigToEnv() {
|
|
521
|
+
const cfg = loadGlobalConfig();
|
|
522
|
+
if (cfg.search?.brave_api_key && !process.env.BRAVE_API_KEY) {
|
|
523
|
+
process.env.BRAVE_API_KEY = cfg.search.brave_api_key;
|
|
524
|
+
}
|
|
525
|
+
if (cfg.search?.tavily_api_key && !process.env.TAVILY_API_KEY) {
|
|
526
|
+
process.env.TAVILY_API_KEY = cfg.search.tavily_api_key;
|
|
527
|
+
}
|
|
528
|
+
if (cfg.search?.context7_api_key && !process.env.CONTEXT7_API_KEY) {
|
|
529
|
+
process.env.CONTEXT7_API_KEY = cfg.search.context7_api_key;
|
|
530
|
+
}
|
|
531
|
+
if (cfg.search?.searxng_url && !process.env.SEARXNG_URL) {
|
|
532
|
+
process.env.SEARXNG_URL = cfg.search.searxng_url;
|
|
533
|
+
}
|
|
534
|
+
if (cfg.search?.provider && !process.env.SEARCH_PROVIDER) {
|
|
535
|
+
process.env.SEARCH_PROVIDER = cfg.search.provider;
|
|
536
|
+
}
|
|
537
|
+
}
|
|
538
|
+
applyGlobalConfigToEnv();
|
|
539
|
+
|
|
540
|
+
export function getSearchConfig() {
|
|
541
|
+
const cfg = loadGlobalConfig();
|
|
542
|
+
return {
|
|
543
|
+
provider: process.env.SEARCH_PROVIDER || cfg.search?.provider || "auto",
|
|
544
|
+
brave_api_key: process.env.BRAVE_API_KEY || cfg.search?.brave_api_key || "",
|
|
545
|
+
tavily_api_key: process.env.TAVILY_API_KEY || cfg.search?.tavily_api_key || "",
|
|
546
|
+
context7_api_key: process.env.CONTEXT7_API_KEY || cfg.search?.context7_api_key || "",
|
|
547
|
+
searxng_url: process.env.SEARXNG_URL || cfg.search?.searxng_url || "",
|
|
548
|
+
};
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
const _initSearchConfig = getSearchConfig();
|
|
552
|
+
const SEARCH_PROVIDER = _initSearchConfig.provider;
|
|
553
|
+
export const BRAVE_API_KEY = _initSearchConfig.brave_api_key;
|
|
554
|
+
export const TAVILY_API_KEY = _initSearchConfig.tavily_api_key;
|
|
555
|
+
const CONTEXT7_API_KEY = _initSearchConfig.context7_api_key;
|
|
556
|
+
const SEARXNG_URL = _initSearchConfig.searxng_url;
|
|
557
|
+
|
|
558
|
+
/**
|
|
559
|
+
* Maximum characters returned by web_fetch before boundary-aware truncation.
|
|
560
|
+
* - Unit: characters
|
|
561
|
+
* - Default: 60000
|
|
562
|
+
* - Override: QWEN_MAX_FETCH_CHARS
|
|
563
|
+
* @type {number}
|
|
564
|
+
*/
|
|
565
|
+
export const MAX_FETCH_CHARS = (() => {
|
|
566
|
+
const parsed = parseInt(process.env.QWEN_MAX_FETCH_CHARS, 10);
|
|
567
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : 60_000;
|
|
568
|
+
})();
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
/**
|
|
572
|
+
* Inactivity threshold for engine telemetry before declaring an engine wedge.
|
|
573
|
+
* - Unit: seconds
|
|
574
|
+
* - Default: 120
|
|
575
|
+
* - Override: QWEN_WEDGE_SILENCE_S
|
|
576
|
+
* @type {number}
|
|
577
|
+
*/
|
|
578
|
+
export const WEDGE_STATS_SILENCE_S = process.env.QWEN_WEDGE_SILENCE_S
|
|
579
|
+
? parseInt(process.env.QWEN_WEDGE_SILENCE_S, 10)
|
|
580
|
+
: 120;
|
|
581
|
+
|
|
582
|
+
/**
|
|
583
|
+
* Flag enabling automated restart and recovery of a wedged inference engine.
|
|
584
|
+
* - Default: true
|
|
585
|
+
* - Override: QWEN_AUTO_HEAL ("0" disables)
|
|
586
|
+
* @type {boolean}
|
|
587
|
+
*/
|
|
588
|
+
export const AUTO_HEAL = process.env.QWEN_AUTO_HEAL !== "0";
|
|
589
|
+
|
|
590
|
+
/**
|
|
591
|
+
* Lockfile path for engine auto-heal serialization.
|
|
592
|
+
* @type {string}
|
|
593
|
+
*/
|
|
594
|
+
export const HEAL_LOCK_FILE = path.join(TASK_DIR, ".engine_heal.lock");
|
|
595
|
+
|
|
596
|
+
/**
|
|
597
|
+
* Time-to-live for engine heal lock acquisition.
|
|
598
|
+
* - Unit: milliseconds
|
|
599
|
+
* - Default: 300000 (5 minutes)
|
|
600
|
+
* @type {number}
|
|
601
|
+
*/
|
|
602
|
+
export const HEAL_LOCK_TTL_MS = 5 * 60_000;
|
|
603
|
+
|
|
604
|
+
/**
|
|
605
|
+
* Lockfile path for exclusive engine boot coordination across processes.
|
|
606
|
+
* @type {string}
|
|
607
|
+
*/
|
|
608
|
+
export const ENGINE_BOOT_LOCK_FILE = path.join(TASK_DIR, ".engine_boot.lock");
|
|
609
|
+
|
|
610
|
+
/**
|
|
611
|
+
* Time-to-live for engine boot lock acquisition.
|
|
612
|
+
* - Unit: milliseconds
|
|
613
|
+
* - Default: 480000 (8 minutes)
|
|
614
|
+
* @type {number}
|
|
615
|
+
*/
|
|
616
|
+
export const ENGINE_BOOT_LOCK_TTL_MS = BOOT_TIMEOUT_MS;
|
|
617
|
+
|
|
618
|
+
/**
|
|
619
|
+
* Destination path for background engine startup logs.
|
|
620
|
+
* - Default: "/tmp/mcp_launch_huge.log"
|
|
621
|
+
* - Override: QWEN_LOG_PATH
|
|
622
|
+
* @type {string}
|
|
623
|
+
*/
|
|
624
|
+
export const ENGINE_LOG_PATH = (() => {
|
|
625
|
+
const cfg = loadGlobalConfig();
|
|
626
|
+
return process.env.QWEN_LOG_PATH || cfg.engine_log_path || "/tmp/mcp_launch_huge.log";
|
|
627
|
+
})();
|
|
628
|
+
|
|
629
|
+
/**
|
|
630
|
+
* Counter ledger tracking cumulative engine wedge and restart events.
|
|
631
|
+
* @type {string}
|
|
632
|
+
*/
|
|
633
|
+
export const WEDGE_COUNTER_FILE = path.join(TASK_DIR, ".wedge_counter.json");
|
|
634
|
+
|
|
635
|
+
/**
|
|
636
|
+
* Maximum allowable payload size accepted by the stream proxy.
|
|
637
|
+
* - Unit: bytes
|
|
638
|
+
* - Value: 52428800 (50 MB)
|
|
639
|
+
* @type {number}
|
|
640
|
+
*/
|
|
641
|
+
export const PROXY_MAX_BODY_BYTES = 50 * 1024 * 1024;
|
|
642
|
+
|
|
643
|
+
/**
|
|
644
|
+
* Guard flag permitting tests or management commands to interrupt, probe, or reboot
|
|
645
|
+
* the live vLLM inference engine. Disabled by default to protect running workloads.
|
|
646
|
+
* - Default: false
|
|
647
|
+
* - Override: ALLOW_ENGINE_INTERRUPT ("1" enables)
|
|
648
|
+
* @type {boolean}
|
|
649
|
+
*/
|
|
650
|
+
export const ALLOW_ENGINE_INTERRUPT = process.env.ALLOW_ENGINE_INTERRUPT === "1";
|
|
651
|
+
|
|
652
|
+
/**
|
|
653
|
+
* Base turn ceiling before requiring supervisor lease extension or cooperative landing.
|
|
654
|
+
* - Unit: count
|
|
655
|
+
* - Default: 80
|
|
656
|
+
* - Override: QWEN_BASE_TURN_BUDGET
|
|
657
|
+
* @type {number}
|
|
658
|
+
*/
|
|
659
|
+
const DEFAULT_BASE_TURN_BUDGET = 80;
|
|
660
|
+
export const BASE_TURN_BUDGET = (() => {
|
|
661
|
+
const parsed = parseInt(process.env.QWEN_BASE_TURN_BUDGET, 10);
|
|
662
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_BASE_TURN_BUDGET;
|
|
663
|
+
})();
|
|
664
|
+
|
|
665
|
+
/**
|
|
666
|
+
* Maximum elastic turn ceiling reachable through supervisor lease extensions.
|
|
667
|
+
* - Unit: count
|
|
668
|
+
* - Default: 200
|
|
669
|
+
* - Override: QWEN_MAX_ELASTIC_TURNS
|
|
670
|
+
* @type {number}
|
|
671
|
+
*/
|
|
672
|
+
const DEFAULT_MAX_ELASTIC_TURNS = 200;
|
|
673
|
+
export const MAX_ELASTIC_TURNS = (() => {
|
|
674
|
+
const parsed = parseInt(process.env.QWEN_MAX_ELASTIC_TURNS, 10);
|
|
675
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_MAX_ELASTIC_TURNS;
|
|
676
|
+
})();
|
|
677
|
+
|
|
678
|
+
/**
|
|
679
|
+
* Maximum allowable GPU KV cache utilization percentage before disallowing lease extension.
|
|
680
|
+
* - Unit: percentage (0-100)
|
|
681
|
+
* - Default: 85.0
|
|
682
|
+
* - Override: QWEN_KV_CACHE_HEADROOM_CEILING
|
|
683
|
+
* @type {number}
|
|
684
|
+
*/
|
|
685
|
+
const DEFAULT_KV_CACHE_HEADROOM_CEILING = 85.0;
|
|
686
|
+
export const KV_CACHE_HEADROOM_CEILING = (() => {
|
|
687
|
+
const parsed = parseFloat(process.env.QWEN_KV_CACHE_HEADROOM_CEILING);
|
|
688
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_KV_CACHE_HEADROOM_CEILING;
|
|
689
|
+
})();
|
|
690
|
+
|
|
691
|
+
/**
|
|
692
|
+
* Minimum average speculative decoding acceptance length required for lease extension.
|
|
693
|
+
* - Unit: tokens per draft step
|
|
694
|
+
* - Default: 2.5
|
|
695
|
+
* - Override: QWEN_SPEC_ACCEPTANCE_FLOOR
|
|
696
|
+
* @type {number}
|
|
697
|
+
*/
|
|
698
|
+
const DEFAULT_SPEC_ACCEPTANCE_FLOOR = 2.5;
|
|
699
|
+
export const SPEC_ACCEPTANCE_FLOOR = (() => {
|
|
700
|
+
const parsed = parseFloat(process.env.QWEN_SPEC_ACCEPTANCE_FLOOR);
|
|
701
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_SPEC_ACCEPTANCE_FLOOR;
|
|
702
|
+
})();
|
|
703
|
+
|
|
704
|
+
/**
|
|
705
|
+
* Sliding window size for action-hash stagnation and loop detection.
|
|
706
|
+
* - Unit: count
|
|
707
|
+
* - Default: 6
|
|
708
|
+
* - Override: QWEN_LOOP_DETECTION_WINDOW
|
|
709
|
+
* @type {number}
|
|
710
|
+
*/
|
|
711
|
+
export const DEFAULT_LOOP_DETECTION_WINDOW = 6;
|
|
712
|
+
export const LOOP_DETECTION_WINDOW = (() => {
|
|
713
|
+
const parsed = parseInt(process.env.QWEN_LOOP_DETECTION_WINDOW, 10);
|
|
714
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_LOOP_DETECTION_WINDOW;
|
|
715
|
+
})();
|
|
716
|
+
|
|
717
|
+
/**
|
|
718
|
+
* Threshold of repeated identical non-mutating actions within the detection window
|
|
719
|
+
* required to trigger loop detection circuit-breaking.
|
|
720
|
+
* - Unit: count
|
|
721
|
+
* - Default: 3
|
|
722
|
+
* - Override: QWEN_LOOP_DETECTION_REPETITIONS
|
|
723
|
+
* @type {number}
|
|
724
|
+
*/
|
|
725
|
+
export const DEFAULT_LOOP_DETECTION_REPETITIONS = 3;
|
|
726
|
+
export const LOOP_DETECTION_REPETITIONS = (() => {
|
|
727
|
+
const parsed = parseInt(process.env.QWEN_LOOP_DETECTION_REPETITIONS, 10);
|
|
728
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_LOOP_DETECTION_REPETITIONS;
|
|
729
|
+
})();
|
|
730
|
+
|
|
731
|
+
/**
|
|
732
|
+
* Character length of recent activity summaries returned in status and wait telemetry endpoints.
|
|
733
|
+
* - Unit: characters
|
|
734
|
+
* - Default: 300
|
|
735
|
+
* - Override: QWEN_SUPERVISOR_PREVIEW_CHARS
|
|
736
|
+
* @type {number}
|
|
737
|
+
*/
|
|
738
|
+
const DEFAULT_SUPERVISOR_PREVIEW_CHARS = 300;
|
|
739
|
+
export const SUPERVISOR_PREVIEW_CHARS = (() => {
|
|
740
|
+
const parsed = parseInt(process.env.QWEN_SUPERVISOR_PREVIEW_CHARS, 10);
|
|
741
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_SUPERVISOR_PREVIEW_CHARS;
|
|
742
|
+
})();
|
|
743
|
+
|
|
744
|
+
/**
|
|
745
|
+
* Optional hard upper bound on session turn count; null allows unbounded orchestrator steering.
|
|
746
|
+
* - Unit: count
|
|
747
|
+
* - Default: null
|
|
748
|
+
* - Override: QWEN_MAX_TURNS
|
|
749
|
+
* @type {number|null}
|
|
750
|
+
*/
|
|
751
|
+
export const MAX_TURNS = process.env.QWEN_MAX_TURNS
|
|
752
|
+
? parseInt(process.env.QWEN_MAX_TURNS, 10)
|
|
753
|
+
: null;
|
|
754
|
+
|
|
755
|
+
/**
|
|
756
|
+
* Maximum re-prompt attempts following a token-ceiling (finish_reason: "length") cutoff.
|
|
757
|
+
* - Unit: count
|
|
758
|
+
* - Default: 8
|
|
759
|
+
* - Override: QWEN_MAX_CONTINUATION_TURNS
|
|
760
|
+
* @type {number}
|
|
761
|
+
*/
|
|
762
|
+
const DEFAULT_MAX_CONTINUATION_TURNS = 8;
|
|
763
|
+
export const MAX_CONTINUATION_TURNS = (() => {
|
|
764
|
+
const parsed = parseInt(process.env.QWEN_MAX_CONTINUATION_TURNS, 10);
|
|
765
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_MAX_CONTINUATION_TURNS;
|
|
766
|
+
})();
|
|
767
|
+
|
|
768
|
+
/**
|
|
769
|
+
* Maximum retry attempts when the inference engine returns an empty generation stream.
|
|
770
|
+
* - Unit: count
|
|
771
|
+
* - Default: 2
|
|
772
|
+
* - Override: QWEN_EMPTY_STREAM_RETRIES
|
|
773
|
+
* @type {number}
|
|
774
|
+
*/
|
|
775
|
+
const DEFAULT_EMPTY_STREAM_RETRIES = 2;
|
|
776
|
+
export const EMPTY_STREAM_RETRIES = (() => {
|
|
777
|
+
const parsed = parseInt(process.env.QWEN_EMPTY_STREAM_RETRIES, 10);
|
|
778
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_EMPTY_STREAM_RETRIES;
|
|
779
|
+
})();
|
|
780
|
+
|
|
781
|
+
/**
|
|
782
|
+
* Retry attempts for empty generation streams when prompt characters exceed EMPTY_STREAM_RETRY_DEPTH_CHARS.
|
|
783
|
+
* - Unit: count
|
|
784
|
+
* - Default: 4
|
|
785
|
+
* - Override: QWEN_EMPTY_STREAM_RETRIES_DEEP
|
|
786
|
+
* @type {number}
|
|
787
|
+
*/
|
|
788
|
+
const DEFAULT_EMPTY_STREAM_RETRIES_DEEP = 4;
|
|
789
|
+
export const EMPTY_STREAM_RETRIES_DEEP = (() => {
|
|
790
|
+
const parsed = parseInt(process.env.QWEN_EMPTY_STREAM_RETRIES_DEEP, 10);
|
|
791
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_EMPTY_STREAM_RETRIES_DEEP;
|
|
792
|
+
})();
|
|
793
|
+
|
|
794
|
+
/**
|
|
795
|
+
* Prompt character threshold that activates EMPTY_STREAM_RETRIES_DEEP.
|
|
796
|
+
* - Unit: characters
|
|
797
|
+
* - Default: 525000 (~150k tokens)
|
|
798
|
+
* - Override: QWEN_EMPTY_STREAM_RETRY_DEPTH_CHARS
|
|
799
|
+
* @type {number}
|
|
800
|
+
*/
|
|
801
|
+
const DEFAULT_EMPTY_STREAM_RETRY_DEPTH_CHARS = 525_000;
|
|
802
|
+
export const EMPTY_STREAM_RETRY_DEPTH_CHARS = (() => {
|
|
803
|
+
const parsed = parseInt(process.env.QWEN_EMPTY_STREAM_RETRY_DEPTH_CHARS, 10);
|
|
804
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_EMPTY_STREAM_RETRY_DEPTH_CHARS;
|
|
805
|
+
})();
|
|
806
|
+
|
|
807
|
+
/**
|
|
808
|
+
* Base delay for exponential backoff between empty-stream retries.
|
|
809
|
+
* - Unit: milliseconds
|
|
810
|
+
* - Default: 2000
|
|
811
|
+
* - Override: QWEN_EMPTY_STREAM_RETRY_BACKOFF_BASE_MS
|
|
812
|
+
* @type {number}
|
|
813
|
+
*/
|
|
814
|
+
const DEFAULT_EMPTY_STREAM_RETRY_BACKOFF_BASE_MS = 2000;
|
|
815
|
+
export const EMPTY_STREAM_RETRY_BACKOFF_BASE_MS = (() => {
|
|
816
|
+
const parsed = parseInt(process.env.QWEN_EMPTY_STREAM_RETRY_BACKOFF_BASE_MS, 10);
|
|
817
|
+
return Number.isFinite(parsed) && parsed >= 0 ? parsed : DEFAULT_EMPTY_STREAM_RETRY_BACKOFF_BASE_MS;
|
|
818
|
+
})();
|
|
819
|
+
|
|
820
|
+
const DEFAULT_EMPTY_STREAM_RETRY_BACKOFF_CAP_MS = 30_000;
|
|
821
|
+
export const EMPTY_STREAM_RETRY_BACKOFF_CAP_MS = (() => {
|
|
822
|
+
const parsed = parseInt(process.env.QWEN_EMPTY_STREAM_RETRY_BACKOFF_CAP_MS, 10);
|
|
823
|
+
return Number.isFinite(parsed) && parsed >= 0 ? parsed : DEFAULT_EMPTY_STREAM_RETRY_BACKOFF_CAP_MS;
|
|
824
|
+
})();
|
|
825
|
+
|
|
826
|
+
/**
|
|
827
|
+
* Task prompt character threshold that emits advisory prompt_over_budget telemetry.
|
|
828
|
+
* - Unit: characters
|
|
829
|
+
* - Default: 1500
|
|
830
|
+
* - Override: QWEN_PROMPT_BUDGET_CHARS
|
|
831
|
+
* @type {number}
|
|
832
|
+
*/
|
|
833
|
+
export const DEFAULT_PROMPT_BUDGET_CHARS = 1500;
|
|
834
|
+
export const PROMPT_BUDGET_CHARS = (() => {
|
|
835
|
+
const parsed = parseInt(process.env.QWEN_PROMPT_BUDGET_CHARS, 10);
|
|
836
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_PROMPT_BUDGET_CHARS;
|
|
837
|
+
})();
|
|
838
|
+
|
|
839
|
+
/**
|
|
840
|
+
* Substantive text length required after stripping guard markers to avoid classification as degenerate.
|
|
841
|
+
* - Unit: characters
|
|
842
|
+
* - Default: 200
|
|
843
|
+
* - Override: QWEN_DEGENERATE_FINAL_SUBSTANTIVE_CHARS
|
|
844
|
+
* @type {number}
|
|
845
|
+
*/
|
|
846
|
+
const DEFAULT_DEGENERATE_FINAL_SUBSTANTIVE_CHARS = 200;
|
|
847
|
+
export const DEGENERATE_FINAL_SUBSTANTIVE_CHARS = (() => {
|
|
848
|
+
const parsed = parseInt(process.env.QWEN_DEGENERATE_FINAL_SUBSTANTIVE_CHARS, 10);
|
|
849
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_DEGENERATE_FINAL_SUBSTANTIVE_CHARS;
|
|
850
|
+
})();
|
|
851
|
+
|
|
852
|
+
/**
|
|
853
|
+
* Maximum turn count within which a truncated final response is evaluated for degeneracy.
|
|
854
|
+
* - Unit: count
|
|
855
|
+
* - Default: 3
|
|
856
|
+
* - Override: QWEN_DEGENERATE_FINAL_MAX_TURNS
|
|
857
|
+
* @type {number}
|
|
858
|
+
*/
|
|
859
|
+
const DEFAULT_DEGENERATE_FINAL_MAX_TURNS = 3;
|
|
860
|
+
export const DEGENERATE_FINAL_MAX_TURNS = (() => {
|
|
861
|
+
const parsed = parseInt(process.env.QWEN_DEGENERATE_FINAL_MAX_TURNS, 10);
|
|
862
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_DEGENERATE_FINAL_MAX_TURNS;
|
|
863
|
+
})();
|
|
864
|
+
|
|
865
|
+
/**
|
|
866
|
+
* Consecutive non-mutating shell execution threshold before emitting an advisory warning.
|
|
867
|
+
* - Unit: count
|
|
868
|
+
* - Default: 4
|
|
869
|
+
* - Override: QWEN_PROBE_BUDGET
|
|
870
|
+
* @type {number}
|
|
871
|
+
*/
|
|
872
|
+
const DEFAULT_PROBE_BUDGET = 4;
|
|
873
|
+
export const PROBE_BUDGET = (() => {
|
|
874
|
+
const parsed = parseInt(process.env.QWEN_PROBE_BUDGET, 10);
|
|
875
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_PROBE_BUDGET;
|
|
876
|
+
})();
|
|
877
|
+
|
|
878
|
+
/**
|
|
879
|
+
* Cumulative turn threshold for emitting an advisory session_warning event.
|
|
880
|
+
* - Unit: count
|
|
881
|
+
* - Default: 60
|
|
882
|
+
* - Override: QWEN_SESSION_WARN_TURNS
|
|
883
|
+
* @type {number}
|
|
884
|
+
*/
|
|
885
|
+
const DEFAULT_SESSION_TURNS_WARN = 60;
|
|
886
|
+
export const SESSION_TURNS_WARN = (() => {
|
|
887
|
+
const parsed = parseInt(process.env.QWEN_SESSION_WARN_TURNS, 10);
|
|
888
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_SESSION_TURNS_WARN;
|
|
889
|
+
})();
|
|
890
|
+
|
|
891
|
+
/**
|
|
892
|
+
* Cumulative turn threshold for emitting a session_turn_limit_recommended advisory event.
|
|
893
|
+
* - Unit: count
|
|
894
|
+
* - Default: 80
|
|
895
|
+
* - Override: QWEN_SESSION_RECOMMEND_TURNS
|
|
896
|
+
* @type {number}
|
|
897
|
+
*/
|
|
898
|
+
const DEFAULT_SESSION_TURNS_RECOMMEND = 80;
|
|
899
|
+
export const SESSION_TURNS_RECOMMEND = (() => {
|
|
900
|
+
const parsed = parseInt(process.env.QWEN_SESSION_RECOMMEND_TURNS, 10);
|
|
901
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_SESSION_TURNS_RECOMMEND;
|
|
902
|
+
})();
|
|
903
|
+
|
|
904
|
+
/**
|
|
905
|
+
* Re-prefill prompt token threshold for emitting a context_depth_warning event.
|
|
906
|
+
* - Unit: tokens
|
|
907
|
+
* - Default: 65536
|
|
908
|
+
* - Override: QWEN_CONTEXT_WARN_TOKENS
|
|
909
|
+
* @type {number}
|
|
910
|
+
*/
|
|
911
|
+
const DEFAULT_CONTEXT_WARN_TOKENS = 65536;
|
|
912
|
+
export const CONTEXT_WARN_TOKENS = (() => {
|
|
913
|
+
const parsed = parseInt(process.env.QWEN_CONTEXT_WARN_TOKENS, 10);
|
|
914
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_CONTEXT_WARN_TOKENS;
|
|
915
|
+
})();
|
|
916
|
+
|
|
917
|
+
/**
|
|
918
|
+
* High-watermark context threshold for proactive session rollover recommendations.
|
|
919
|
+
* Emits an advisory recommendation when prompt tokens exceed this threshold.
|
|
920
|
+
* - Unit: tokens
|
|
921
|
+
* - Default: 180000
|
|
922
|
+
* - Override: QWEN_CONTEXT_HIGH_WATERMARK_TOKENS
|
|
923
|
+
* @type {number}
|
|
924
|
+
*/
|
|
925
|
+
const DEFAULT_CONTEXT_HIGH_WATERMARK_TOKENS = 180_000;
|
|
926
|
+
export const CONTEXT_HIGH_WATERMARK_TOKENS = (() => {
|
|
927
|
+
const parsed = parseInt(process.env.QWEN_CONTEXT_HIGH_WATERMARK_TOKENS, 10);
|
|
928
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_CONTEXT_HIGH_WATERMARK_TOKENS;
|
|
929
|
+
})();
|
|
930
|
+
|
|
931
|
+
/**
|
|
932
|
+
* Hard emergency ceiling for context tokens before halting further accumulation.
|
|
933
|
+
* - Unit: tokens
|
|
934
|
+
* - Default: 215000
|
|
935
|
+
* - Override: QWEN_CONTEXT_EMERGENCY_CEILING_TOKENS
|
|
936
|
+
* @type {number}
|
|
937
|
+
*/
|
|
938
|
+
const DEFAULT_CONTEXT_EMERGENCY_CEILING_TOKENS = 215_000;
|
|
939
|
+
export const CONTEXT_EMERGENCY_CEILING_TOKENS = (() => {
|
|
940
|
+
const parsed = parseInt(process.env.QWEN_CONTEXT_EMERGENCY_CEILING_TOKENS, 10);
|
|
941
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_CONTEXT_EMERGENCY_CEILING_TOKENS;
|
|
942
|
+
})();
|
|
943
|
+
|
|
944
|
+
/**
|
|
945
|
+
* Maximum file read size enforced when context high-watermark is active.
|
|
946
|
+
* - Unit: bytes
|
|
947
|
+
* - Default: 16384 (16 KB)
|
|
948
|
+
* - Override: QWEN_READ_GOVERNOR_MAX_BYTES
|
|
949
|
+
* @type {number}
|
|
950
|
+
*/
|
|
951
|
+
const DEFAULT_READ_GOVERNOR_MAX_BYTES = 16 * 1024;
|
|
952
|
+
export const READ_GOVERNOR_MAX_BYTES = (() => {
|
|
953
|
+
const parsed = parseInt(process.env.QWEN_READ_GOVERNOR_MAX_BYTES, 10);
|
|
954
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_READ_GOVERNOR_MAX_BYTES;
|
|
955
|
+
})();
|
|
956
|
+
|
|
957
|
+
/**
|
|
958
|
+
* Tool result size threshold above which payload is spilled to disk.
|
|
959
|
+
* - Unit: bytes
|
|
960
|
+
* - Default: 16384 (16 KB)
|
|
961
|
+
* - Override: QWEN_TOOL_SPILL_BYTES
|
|
962
|
+
* @type {number}
|
|
963
|
+
*/
|
|
964
|
+
const DEFAULT_TOOL_SPILL_BYTES = 16384;
|
|
965
|
+
export const TOOL_SPILL_BYTES = (() => {
|
|
966
|
+
const parsed = parseInt(process.env.QWEN_TOOL_SPILL_BYTES, 10);
|
|
967
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_TOOL_SPILL_BYTES;
|
|
968
|
+
})();
|
|
969
|
+
|
|
970
|
+
/**
|
|
971
|
+
* Generation token limit for bounded findings extraction on deliberation budget exhaustion.
|
|
972
|
+
* - Unit: tokens
|
|
973
|
+
* - Default: 4096
|
|
974
|
+
* - Override: QWEN_SALVAGE_MAX_TOKENS
|
|
975
|
+
* @type {number}
|
|
976
|
+
*/
|
|
977
|
+
const DEFAULT_SALVAGE_MAX_TOKENS = 4096;
|
|
978
|
+
export const SALVAGE_MAX_TOKENS = (() => {
|
|
979
|
+
const parsed = parseInt(process.env.QWEN_SALVAGE_MAX_TOKENS, 10);
|
|
980
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_SALVAGE_MAX_TOKENS;
|
|
981
|
+
})();
|
|
982
|
+
|
|
983
|
+
// ---------------------------------------------------------------------------
|
|
984
|
+
// Engine configuration: model / baseURL / max_context / launch_command
|
|
985
|
+
// Precedence: environment variables > ~/.castor/config.json > built-in defaults.
|
|
986
|
+
// ---------------------------------------------------------------------------
|
|
987
|
+
|
|
988
|
+
const DEFAULT_MODEL = "Qwen3.8-27B";
|
|
989
|
+
const DEFAULT_MAX_CONTEXT = MAX_LEN_HUGE;
|
|
990
|
+
const DEFAULT_LAUNCH_COMMAND = "";
|
|
991
|
+
|
|
992
|
+
/**
|
|
993
|
+
* Resolves the engine configuration with precedence:
|
|
994
|
+
* 1. Environment variables (CASTOR_MODEL/QWEN_MODEL, CASTOR_BASE_URL/QWEN_BASE_URL,
|
|
995
|
+
* CASTOR_MAX_CONTEXT/QWEN_MAX_CONTEXT, CASTOR_LAUNCH_COMMAND/QWEN_LAUNCH_COMMAND,
|
|
996
|
+
* CASTOR_TOOL_PREFIX/MCP_TOOL_PREFIX)
|
|
997
|
+
* 2. Active profile in ~/.castor/config.json (profiles[active_profile])
|
|
998
|
+
* 3. ~/.castor/config.json root keys (model, baseURL, max_context, launch_command, tool_prefix)
|
|
999
|
+
* 4. Built-in defaults
|
|
1000
|
+
*
|
|
1001
|
+
* @returns {{
|
|
1002
|
+
* model: string,
|
|
1003
|
+
* baseURL: string,
|
|
1004
|
+
* max_context: number,
|
|
1005
|
+
* launch_command: string,
|
|
1006
|
+
* tool_prefix: string,
|
|
1007
|
+
* api_key?: string,
|
|
1008
|
+
* engine_type?: string,
|
|
1009
|
+
* stop_command?: string,
|
|
1010
|
+
* engine_log_path?: string
|
|
1011
|
+
* }}
|
|
1012
|
+
*/
|
|
1013
|
+
export function getEngineConfig() {
|
|
1014
|
+
const cfg = loadGlobalConfig();
|
|
1015
|
+
const profile =
|
|
1016
|
+
(cfg.active_profile && cfg.profiles && cfg.profiles[cfg.active_profile]) ||
|
|
1017
|
+
{};
|
|
1018
|
+
const merged = { ...cfg, ...profile };
|
|
1019
|
+
|
|
1020
|
+
return {
|
|
1021
|
+
model:
|
|
1022
|
+
process.env.CASTOR_MODEL ||
|
|
1023
|
+
process.env.QWEN_MODEL ||
|
|
1024
|
+
merged.model ||
|
|
1025
|
+
DEFAULT_MODEL,
|
|
1026
|
+
baseURL:
|
|
1027
|
+
process.env.CASTOR_BASE_URL ||
|
|
1028
|
+
process.env.QWEN_BASE_URL ||
|
|
1029
|
+
merged.baseURL ||
|
|
1030
|
+
BASE_URL,
|
|
1031
|
+
max_context:
|
|
1032
|
+
parseInt(
|
|
1033
|
+
process.env.CASTOR_MAX_CONTEXT || process.env.QWEN_MAX_CONTEXT,
|
|
1034
|
+
10
|
|
1035
|
+
) ||
|
|
1036
|
+
merged.max_context ||
|
|
1037
|
+
DEFAULT_MAX_CONTEXT,
|
|
1038
|
+
launch_command:
|
|
1039
|
+
process.env.CASTOR_LAUNCH_COMMAND ||
|
|
1040
|
+
process.env.QWEN_LAUNCH_COMMAND ||
|
|
1041
|
+
merged.launch_command ||
|
|
1042
|
+
DEFAULT_LAUNCH_COMMAND,
|
|
1043
|
+
tool_prefix:
|
|
1044
|
+
process.env.CASTOR_TOOL_PREFIX ??
|
|
1045
|
+
process.env.MCP_TOOL_PREFIX ??
|
|
1046
|
+
merged.tool_prefix ??
|
|
1047
|
+
"castor",
|
|
1048
|
+
api_key:
|
|
1049
|
+
process.env.CASTOR_API_KEY ||
|
|
1050
|
+
process.env.QWEN_API_KEY ||
|
|
1051
|
+
merged.api_key ||
|
|
1052
|
+
"",
|
|
1053
|
+
engine_type:
|
|
1054
|
+
process.env.CASTOR_ENGINE_TYPE ||
|
|
1055
|
+
process.env.QWEN_ENGINE_TYPE ||
|
|
1056
|
+
merged.engine_type ||
|
|
1057
|
+
"vllm",
|
|
1058
|
+
stop_command:
|
|
1059
|
+
process.env.CASTOR_STOP_COMMAND ||
|
|
1060
|
+
process.env.QWEN_STOP_COMMAND ||
|
|
1061
|
+
merged.stop_command ||
|
|
1062
|
+
"",
|
|
1063
|
+
engine_log_path:
|
|
1064
|
+
process.env.CASTOR_LOG_PATH ||
|
|
1065
|
+
process.env.QWEN_LOG_PATH ||
|
|
1066
|
+
merged.engine_log_path ||
|
|
1067
|
+
"/tmp/mcp_launch_huge.log",
|
|
1068
|
+
};
|
|
1069
|
+
}
|
|
1070
|
+
|
|
1071
|
+
/**
|
|
1072
|
+
* Model name served by the local engine.
|
|
1073
|
+
* - Default: "Qwen3.8-27B"
|
|
1074
|
+
* - Override: CASTOR_MODEL / QWEN_MODEL (env) or `model` in ~/.castor/config.json
|
|
1075
|
+
* @type {string}
|
|
1076
|
+
*/
|
|
1077
|
+
export const MODEL = getEngineConfig().model;
|
|
1078
|
+
|
|
1079
|
+
/**
|
|
1080
|
+
* Nominal maximum context window for the served model, in tokens.
|
|
1081
|
+
* - Default: 245760
|
|
1082
|
+
* - Override: CASTOR_MAX_CONTEXT / QWEN_MAX_CONTEXT (env) or `max_context` in ~/.castor/config.json
|
|
1083
|
+
* @type {number}
|
|
1084
|
+
*/
|
|
1085
|
+
export const MAX_CONTEXT = getEngineConfig().max_context;
|
|
1086
|
+
|
|
1087
|
+
/**
|
|
1088
|
+
* Shell command used to (re)start the inference engine.
|
|
1089
|
+
* - Default: "" (no managed launch)
|
|
1090
|
+
* - Override: CASTOR_LAUNCH_COMMAND / QWEN_LAUNCH_COMMAND (env) or `launch_command` in ~/.castor/config.json
|
|
1091
|
+
* @type {string}
|
|
1092
|
+
*/
|
|
1093
|
+
export const LAUNCH_COMMAND = getEngineConfig().launch_command;
|
|
1094
|
+
|
|
1095
|
+
/**
|
|
1096
|
+
* Prefix applied to registered MCP tool names (e.g. "castor" -> "castor_coworker").
|
|
1097
|
+
* - Default: "castor"
|
|
1098
|
+
* - Override: CASTOR_TOOL_PREFIX / MCP_TOOL_PREFIX (env) or `tool_prefix` in ~/.castor/config.json
|
|
1099
|
+
* @type {string}
|
|
1100
|
+
*/
|
|
1101
|
+
export const TOOL_PREFIX = getEngineConfig().tool_prefix;
|
|
1102
|
+
|
|
1103
|
+
// ---------------------------------------------------------------------------
|
|
1104
|
+
// Config-file-aware resolved values.
|
|
1105
|
+
//
|
|
1106
|
+
// These are the "effective" values that honor ~/.castor/config.json in addition
|
|
1107
|
+
// to environment variables. They are defined at the bottom of the file (after
|
|
1108
|
+
// GLOBAL_CONFIG_FILE / GLOBAL_ENV_FILE are initialized) so that calling
|
|
1109
|
+
// loadGlobalConfig() here is safe (no TDZ).
|
|
1110
|
+
//
|
|
1111
|
+
// Precedence: environment variable > config.json > built-in default.
|
|
1112
|
+
// ---------------------------------------------------------------------------
|
|
1113
|
+
|
|
1114
|
+
/**
|
|
1115
|
+
* Effective stream proxy port (env > config.json > default 18022).
|
|
1116
|
+
* @type {number}
|
|
1117
|
+
*/
|
|
1118
|
+
export const STREAM_PROXY_PORT_RESOLVED = (() => {
|
|
1119
|
+
const envPort = process.env.STREAM_PROXY_PORT || process.env.VLLM_PROXY_PORT;
|
|
1120
|
+
if (envPort) {
|
|
1121
|
+
const p = parseInt(envPort, 10);
|
|
1122
|
+
if (Number.isFinite(p) && p > 0) return p;
|
|
1123
|
+
}
|
|
1124
|
+
const cfg = loadGlobalConfig();
|
|
1125
|
+
if (typeof cfg.stream_proxy_port === "number" && cfg.stream_proxy_port > 0) {
|
|
1126
|
+
return cfg.stream_proxy_port;
|
|
1127
|
+
}
|
|
1128
|
+
return 18022;
|
|
1129
|
+
})();
|
|
1130
|
+
|
|
1131
|
+
/**
|
|
1132
|
+
* Effective use-stream-proxy flag (env > config.json > default true).
|
|
1133
|
+
* @type {boolean}
|
|
1134
|
+
*/
|
|
1135
|
+
export const USE_STREAM_PROXY_RESOLVED = (() => {
|
|
1136
|
+
if (process.env.USE_STREAM_PROXY !== undefined) {
|
|
1137
|
+
return process.env.USE_STREAM_PROXY !== "false" && process.env.USE_STREAM_PROXY !== "0";
|
|
1138
|
+
}
|
|
1139
|
+
const cfg = loadGlobalConfig();
|
|
1140
|
+
if (typeof cfg.use_stream_proxy === "boolean") return cfg.use_stream_proxy;
|
|
1141
|
+
return true;
|
|
1142
|
+
})();
|
|
1143
|
+
|
|
1144
|
+
/**
|
|
1145
|
+
* Effective status port (env > config.json > default 18021).
|
|
1146
|
+
* @type {number}
|
|
1147
|
+
*/
|
|
1148
|
+
export const STATUS_PORT_RESOLVED = (() => {
|
|
1149
|
+
const envPort = process.env.STATUS_PORT;
|
|
1150
|
+
if (envPort) {
|
|
1151
|
+
const p = parseInt(envPort, 10);
|
|
1152
|
+
if (Number.isFinite(p) && p > 0) return p;
|
|
1153
|
+
}
|
|
1154
|
+
const cfg = loadGlobalConfig();
|
|
1155
|
+
if (typeof cfg.status_port === "number" && cfg.status_port > 0) return cfg.status_port;
|
|
1156
|
+
return 18021;
|
|
1157
|
+
})();
|
|
1158
|
+
|
|
1159
|
+
/**
|
|
1160
|
+
* Effective vLLM port (env > config.json > default 18020).
|
|
1161
|
+
* @type {number}
|
|
1162
|
+
*/
|
|
1163
|
+
export const VLLM_PORT_RESOLVED = (() => {
|
|
1164
|
+
const envPort = process.env.VLLM_PORT;
|
|
1165
|
+
if (envPort) {
|
|
1166
|
+
const p = parseInt(envPort, 10);
|
|
1167
|
+
if (Number.isFinite(p) && p > 0) return p;
|
|
1168
|
+
}
|
|
1169
|
+
const cfg = loadGlobalConfig();
|
|
1170
|
+
if (typeof cfg.vllm_port === "number" && cfg.vllm_port > 0) return cfg.vllm_port;
|
|
1171
|
+
return 18020;
|
|
1172
|
+
})();
|
|
1173
|
+
|
|
1174
|
+
/**
|
|
1175
|
+
* Engine API key (env > config.json > empty string).
|
|
1176
|
+
* @type {string}
|
|
1177
|
+
*/
|
|
1178
|
+
export const API_KEY = (() => {
|
|
1179
|
+
const cfg = loadGlobalConfig();
|
|
1180
|
+
return process.env.QWEN_API_KEY || cfg.api_key || "";
|
|
1181
|
+
})();
|
|
1182
|
+
|
|
1183
|
+
/**
|
|
1184
|
+
* Engine type tag (env > config.json > "vllm").
|
|
1185
|
+
* @type {string}
|
|
1186
|
+
*/
|
|
1187
|
+
export const ENGINE_TYPE = (() => {
|
|
1188
|
+
const cfg = loadGlobalConfig();
|
|
1189
|
+
return process.env.QWEN_ENGINE_TYPE || cfg.engine_type || "vllm";
|
|
1190
|
+
})();
|
|
1191
|
+
|
|
1192
|
+
/**
|
|
1193
|
+
* Shell command used to stop the inference engine (env > config.json > "").
|
|
1194
|
+
* @type {string}
|
|
1195
|
+
*/
|
|
1196
|
+
export const STOP_COMMAND = (() => {
|
|
1197
|
+
const cfg = loadGlobalConfig();
|
|
1198
|
+
return process.env.QWEN_STOP_COMMAND || cfg.stop_command || "";
|
|
1199
|
+
})();
|
|
1200
|
+
|
|
1201
|
+
|
|
1202
|
+
|
|
1203
|
+
|
|
1204
|
+
|