mcp-castor 2026.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +487 -0
- package/bin/castor.js +706 -0
- package/index.js +206 -0
- package/package.json +97 -0
- package/skills/canary-test-staging/SKILL.md +24 -0
- package/skills/evo-mutation-rollback/SKILL.md +29 -0
- package/skills/hypothesis-generation/SKILL.md +26 -0
- package/skills/traceback-condensing/SKILL.md +26 -0
- package/src/castor_runner.js +469 -0
- package/src/config.js +1204 -0
- package/src/env.js +10 -0
- package/src/evo_engine.js +214 -0
- package/src/harness/core/events.js +75 -0
- package/src/harness/core/kernel.js +209 -0
- package/src/harness/evo/evaluator.js +156 -0
- package/src/harness/evo/evo_operator.js +550 -0
- package/src/harness/evo/lineage_dag.js +383 -0
- package/src/harness/evo/trace_repair.js +173 -0
- package/src/harness/evo/watchdog.js +72 -0
- package/src/harness/loop_detector.js +135 -0
- package/src/harness/runner.js +1216 -0
- package/src/harness/services/ast_service.js +1813 -0
- package/src/harness/services/event_logger.js +275 -0
- package/src/harness/services/mcp_bridge.js +408 -0
- package/src/harness/services/provider_vllm.js +728 -0
- package/src/harness/services/sandbox_fs.js +1238 -0
- package/src/harness/services/searxng_lifecycle.js +254 -0
- package/src/harness/services/shell_executor.js +264 -0
- package/src/harness/services/shell_validator.js +506 -0
- package/src/harness/services/web_service.js +828 -0
- package/src/platform.js +344 -0
- package/src/repetition_detector.js +139 -0
- package/src/semaphore.js +373 -0
- package/src/server_lifecycle.js +781 -0
- package/src/skills.js +400 -0
- package/src/state_pruner.js +392 -0
- package/src/task_registry.js +1357 -0
- package/src/telemetry.js +638 -0
- package/src/tools.js +997 -0
- package/src/wsl_bridge.js +629 -0
- package/src/wsl_env.js +171 -0
- package/stream_proxy.js +453 -0
package/src/telemetry.js
ADDED
|
@@ -0,0 +1,638 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
import { QWEN_STATE_DIR, IS_WINDOWS, ENGINE_LOG_PATH } from "./config.js";
|
|
4
|
+
|
|
5
|
+
const TELEMETRY_DIR = path.join(QWEN_STATE_DIR, "telemetry");
|
|
6
|
+
const STATS_FILE = path.join(TELEMETRY_DIR, "stats.json");
|
|
7
|
+
|
|
8
|
+
// Frontier commercial baseline pricing (Claude Sonnet 5 tier: $2.00/M prompt, $10.00/M completion)
|
|
9
|
+
const PROMPT_COST_PER_MILLION = 2.0;
|
|
10
|
+
const COMPLETION_COST_PER_MILLION = 10.0;
|
|
11
|
+
const BENCHMARK_MODEL = "Claude Sonnet 5";
|
|
12
|
+
|
|
13
|
+
// Authoritative September 2026 frontier model reference rates
|
|
14
|
+
const FRONTIER_BENCHMARKS = {
|
|
15
|
+
"Claude Sonnet 5": { promptPerM: 2.0, compPerM: 10.0, context: "500K" },
|
|
16
|
+
"Claude Opus 5.5": { promptPerM: 4.0, compPerM: 20.0, context: "1,000K" },
|
|
17
|
+
"Claude Fable 5.1": { promptPerM: 10.0, compPerM: 50.0, context: "1,000K" },
|
|
18
|
+
"GPT-6 Astra": { promptPerM: 10.0, compPerM: 50.0, context: "1,050K" },
|
|
19
|
+
"GLM-5.3": { promptPerM: 1.4, compPerM: 4.4, context: "200K" },
|
|
20
|
+
};
|
|
21
|
+
|
|
22
|
+
export function calculateCostSaved(promptTokens, completionTokens, promptRate = PROMPT_COST_PER_MILLION, compRate = COMPLETION_COST_PER_MILLION) {
|
|
23
|
+
const promptCost = (promptTokens / 1_000_000) * promptRate;
|
|
24
|
+
const compCost = (completionTokens / 1_000_000) * compRate;
|
|
25
|
+
return Number((promptCost + compCost).toFixed(2));
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** Baseline aggregated statistics across historical sessions. */
|
|
29
|
+
export const DEFAULT_STATS = {
|
|
30
|
+
total_completion_tokens: 22_045_497,
|
|
31
|
+
total_reasoning_tokens: 23_832_473,
|
|
32
|
+
total_prompt_tokens_measured: 615_423_782,
|
|
33
|
+
total_prompt_tokens_estimated: 694_147_295,
|
|
34
|
+
total_prompt_tokens: 1_309_571_077,
|
|
35
|
+
total_turns: 17_616,
|
|
36
|
+
total_sessions: 453,
|
|
37
|
+
total_tasks_completed: 377,
|
|
38
|
+
total_tasks_failed: 82,
|
|
39
|
+
total_tasks_cancelled: 15,
|
|
40
|
+
avg_ttft_ms: 20613.7,
|
|
41
|
+
avg_prefill_ms: 20613.7,
|
|
42
|
+
avg_generation_ms: 13995.3,
|
|
43
|
+
avg_decode_tps: 61.14,
|
|
44
|
+
avg_prefill_tps: 15032.79,
|
|
45
|
+
avg_tpot_ms: 18.16,
|
|
46
|
+
reasoning_effort: {
|
|
47
|
+
xhigh: 116,
|
|
48
|
+
medium: 6838,
|
|
49
|
+
low: 699,
|
|
50
|
+
},
|
|
51
|
+
vllm_engine_metrics: {
|
|
52
|
+
prefix_cache_hit_rate_pct: 92.6,
|
|
53
|
+
peak_gpu_kv_cache_pct: 99.4,
|
|
54
|
+
spec_mean_acceptance_length: 5.33,
|
|
55
|
+
spec_draft_acceptance_rate_pct: 61.9,
|
|
56
|
+
active_generation_throughput_tps: 44.54,
|
|
57
|
+
active_prompt_throughput_tps: 1376.75,
|
|
58
|
+
},
|
|
59
|
+
total_tool_calls: 23_965,
|
|
60
|
+
total_tool_errors: 937,
|
|
61
|
+
tool_calls: {
|
|
62
|
+
bash: 11809,
|
|
63
|
+
read_file: 5173,
|
|
64
|
+
edit_file: 2898,
|
|
65
|
+
write_file: 1161,
|
|
66
|
+
list_dir: 601,
|
|
67
|
+
search_code: 1266,
|
|
68
|
+
evo_propose_candidate: 135,
|
|
69
|
+
evo_evaluate_candidate: 117,
|
|
70
|
+
evo_select_candidate: 114,
|
|
71
|
+
ext_context7_mcp_query_docs: 45,
|
|
72
|
+
evo_status: 25,
|
|
73
|
+
ext_context7_mcp_resolve_library_id: 18,
|
|
74
|
+
avo_propose_candidate: 15,
|
|
75
|
+
avo_select_candidate: 10,
|
|
76
|
+
avo_evaluate_candidate: 5,
|
|
77
|
+
ast_search: 9,
|
|
78
|
+
exec_command: 1,
|
|
79
|
+
avo_status: 1,
|
|
80
|
+
evo_revert_candidate: 1,
|
|
81
|
+
apply_patch: 4,
|
|
82
|
+
web_search: 328,
|
|
83
|
+
web_fetch: 229,
|
|
84
|
+
},
|
|
85
|
+
tool_errors: {
|
|
86
|
+
read_file: 598,
|
|
87
|
+
edit_file: 139,
|
|
88
|
+
write_file: 49,
|
|
89
|
+
bash: 17,
|
|
90
|
+
list_dir: 14,
|
|
91
|
+
search_code: 27,
|
|
92
|
+
evo_evaluate_candidate: 3,
|
|
93
|
+
ast_search: 2,
|
|
94
|
+
apply_patch: 4,
|
|
95
|
+
web_fetch: 22,
|
|
96
|
+
web_search: 62,
|
|
97
|
+
},
|
|
98
|
+
benchmark_model: BENCHMARK_MODEL,
|
|
99
|
+
estimated_cost_saved_usd: 2839.6,
|
|
100
|
+
first_recorded_session: "2026-09-05T03:58:07.228Z",
|
|
101
|
+
last_recorded_session: "2026-09-24T18:10:46.123Z",
|
|
102
|
+
history_ingested: true,
|
|
103
|
+
last_updated: new Date().toISOString(),
|
|
104
|
+
};
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Loads the current cumulative stats from disk or initializes defaults.
|
|
108
|
+
*/
|
|
109
|
+
export function getCumulativeTelemetry() {
|
|
110
|
+
try {
|
|
111
|
+
if (fs.existsSync(STATS_FILE)) {
|
|
112
|
+
const data = JSON.parse(fs.readFileSync(STATS_FILE, "utf8"));
|
|
113
|
+
return {
|
|
114
|
+
...DEFAULT_STATS,
|
|
115
|
+
...data,
|
|
116
|
+
reasoning_effort: {
|
|
117
|
+
...DEFAULT_STATS.reasoning_effort,
|
|
118
|
+
...(data.reasoning_effort || {}),
|
|
119
|
+
},
|
|
120
|
+
vllm_engine_metrics: {
|
|
121
|
+
...DEFAULT_STATS.vllm_engine_metrics,
|
|
122
|
+
...(data.vllm_engine_metrics || {}),
|
|
123
|
+
},
|
|
124
|
+
tool_calls: {
|
|
125
|
+
...DEFAULT_STATS.tool_calls,
|
|
126
|
+
...(data.tool_calls || {}),
|
|
127
|
+
},
|
|
128
|
+
tool_errors: {
|
|
129
|
+
...DEFAULT_STATS.tool_errors,
|
|
130
|
+
...(data.tool_errors || {}),
|
|
131
|
+
},
|
|
132
|
+
};
|
|
133
|
+
}
|
|
134
|
+
} catch (err) {
|
|
135
|
+
// C1: corrupt stats file → quarantine, never silent-zero.
|
|
136
|
+
const quarantineName = `stats.json.corrupt-${Date.now()}`;
|
|
137
|
+
try {
|
|
138
|
+
fs.renameSync(STATS_FILE, path.join(TELEMETRY_DIR, quarantineName));
|
|
139
|
+
} catch {}
|
|
140
|
+
process.stderr.write(`[Telemetry] Corrupt stats file quarantined to ${quarantineName}: ${err.message}\n`);
|
|
141
|
+
}
|
|
142
|
+
return JSON.parse(JSON.stringify(DEFAULT_STATS));
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Persists updated cumulative stats atomically across Windows & WSL.
|
|
147
|
+
*/
|
|
148
|
+
export function saveCumulativeTelemetry(stats) {
|
|
149
|
+
try {
|
|
150
|
+
if (!fs.existsSync(TELEMETRY_DIR)) {
|
|
151
|
+
fs.mkdirSync(TELEMETRY_DIR, { recursive: true });
|
|
152
|
+
}
|
|
153
|
+
stats.estimated_cost_saved_usd = calculateCostSaved(
|
|
154
|
+
stats.total_prompt_tokens,
|
|
155
|
+
stats.total_completion_tokens
|
|
156
|
+
);
|
|
157
|
+
stats.last_updated = new Date().toISOString();
|
|
158
|
+
const tmp = `${STATS_FILE}.tmp.${process.pid}.${Date.now()}`;
|
|
159
|
+
fs.writeFileSync(tmp, JSON.stringify(stats, null, 2), "utf8");
|
|
160
|
+
fs.renameSync(tmp, STATS_FILE);
|
|
161
|
+
} catch (err) {
|
|
162
|
+
console.error("[Telemetry] Failed to persist stats:", err.message);
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Records telemetry from an executed turn.
|
|
168
|
+
* @param {object} params
|
|
169
|
+
* @param {number} [params.completionTokens=0] - Tokens generated during the completion phase.
|
|
170
|
+
* @param {number} [params.promptTokens=0] - Tokens processed during the prefill phase.
|
|
171
|
+
* @param {number} [params.reasoningTokens=0] - Deliberative reasoning tokens.
|
|
172
|
+
* @param {number} [params.ttftMs=null] - Time to first token in milliseconds.
|
|
173
|
+
* @param {number} [params.prefillMs=null] - Duration of the prefill phase in milliseconds.
|
|
174
|
+
* @param {number} [params.generationMs=null] - Duration of the generation phase in milliseconds.
|
|
175
|
+
* @param {number} [params.totalMs=null] - Total turn duration in milliseconds.
|
|
176
|
+
* @param {string} [params.effort="medium"] - Reasoning effort tier.
|
|
177
|
+
*/
|
|
178
|
+
export function recordTurnTelemetry({
|
|
179
|
+
completionTokens = 0,
|
|
180
|
+
promptTokens = 0,
|
|
181
|
+
reasoningTokens = 0,
|
|
182
|
+
ttftMs = null,
|
|
183
|
+
prefillMs = null,
|
|
184
|
+
generationMs = null,
|
|
185
|
+
totalMs = null,
|
|
186
|
+
effort = "medium",
|
|
187
|
+
decodeTps = null,
|
|
188
|
+
prefillTps = null,
|
|
189
|
+
tpotMs = null,
|
|
190
|
+
} = {}) {
|
|
191
|
+
const stats = getCumulativeTelemetry();
|
|
192
|
+
stats.total_completion_tokens += completionTokens;
|
|
193
|
+
stats.total_reasoning_tokens += reasoningTokens;
|
|
194
|
+
stats.total_prompt_tokens_measured += promptTokens;
|
|
195
|
+
stats.total_prompt_tokens += promptTokens;
|
|
196
|
+
stats.total_turns += 1;
|
|
197
|
+
|
|
198
|
+
if (stats.reasoning_effort[effort] !== undefined) {
|
|
199
|
+
stats.reasoning_effort[effort] += 1;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
const effectivePrefillMs = prefillMs ?? ttftMs;
|
|
203
|
+
if (typeof effectivePrefillMs === "number" && effectivePrefillMs > 0) {
|
|
204
|
+
stats.avg_ttft_ms = Number(
|
|
205
|
+
(((stats.avg_ttft_ms || effectivePrefillMs) * 0.95) + (effectivePrefillMs * 0.05)).toFixed(1)
|
|
206
|
+
);
|
|
207
|
+
stats.avg_prefill_ms = stats.avg_ttft_ms;
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
const effectiveGenMs =
|
|
211
|
+
typeof generationMs === "number" && generationMs >= 0
|
|
212
|
+
? generationMs
|
|
213
|
+
: (typeof totalMs === "number" && typeof effectivePrefillMs === "number"
|
|
214
|
+
? Math.max(0, totalMs - effectivePrefillMs)
|
|
215
|
+
: null);
|
|
216
|
+
|
|
217
|
+
if (typeof effectiveGenMs === "number" && effectiveGenMs >= 0) {
|
|
218
|
+
stats.avg_generation_ms = Number(
|
|
219
|
+
(((stats.avg_generation_ms || effectiveGenMs) * 0.95) + (effectiveGenMs * 0.05)).toFixed(1)
|
|
220
|
+
);
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
// Running averages for the standard throughput / per-token-latency rates.
|
|
224
|
+
// Same 0.95/0.05 EMA convention as the latency averages above; only updated
|
|
225
|
+
// when the provider actually produced a finite, non-negative value (null
|
|
226
|
+
// rates from degenerate turns are ignored, never averaged in).
|
|
227
|
+
if (typeof decodeTps === "number" && decodeTps >= 0) {
|
|
228
|
+
stats.avg_decode_tps = Number(
|
|
229
|
+
(((stats.avg_decode_tps || decodeTps) * 0.95) + (decodeTps * 0.05)).toFixed(2)
|
|
230
|
+
);
|
|
231
|
+
}
|
|
232
|
+
if (typeof prefillTps === "number" && prefillTps >= 0) {
|
|
233
|
+
stats.avg_prefill_tps = Number(
|
|
234
|
+
(((stats.avg_prefill_tps || prefillTps) * 0.95) + (prefillTps * 0.05)).toFixed(2)
|
|
235
|
+
);
|
|
236
|
+
}
|
|
237
|
+
if (typeof tpotMs === "number" && tpotMs >= 0) {
|
|
238
|
+
stats.avg_tpot_ms = Number(
|
|
239
|
+
(((stats.avg_tpot_ms || tpotMs) * 0.95) + (tpotMs * 0.05)).toFixed(2)
|
|
240
|
+
);
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
saveCumulativeTelemetry(stats);
|
|
244
|
+
|
|
245
|
+
// Append a per-turn record to the rolling turns.jsonl ledger so that
|
|
246
|
+
// time-sliced queries (queryTelemetry) can reconstruct distributions and
|
|
247
|
+
// rates over arbitrary windows. Capped at MAX_TURNS_LEDGER lines; when the
|
|
248
|
+
// cap is exceeded the oldest lines are dropped (FIFO) to bound disk growth.
|
|
249
|
+
appendTurnRecord({
|
|
250
|
+
ts: Date.now(),
|
|
251
|
+
prompt: promptTokens,
|
|
252
|
+
comp: completionTokens,
|
|
253
|
+
reasoning: reasoningTokens,
|
|
254
|
+
ttft: effectivePrefillMs,
|
|
255
|
+
gen: effectiveGenMs,
|
|
256
|
+
total: totalMs,
|
|
257
|
+
decodeTps,
|
|
258
|
+
prefillTps,
|
|
259
|
+
tpotMs,
|
|
260
|
+
effort,
|
|
261
|
+
});
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
const TURNS_LEDGER_FILE = () => path.join(TELEMETRY_DIR, "turns.jsonl");
|
|
265
|
+
const MAX_TURNS_LEDGER = 20000;
|
|
266
|
+
|
|
267
|
+
/**
|
|
268
|
+
* Appends a single per-turn record to the rolling turns.jsonl ledger,
|
|
269
|
+
* trimming the oldest lines when the cap is exceeded. Never throws: a
|
|
270
|
+
* telemetry-write failure must not break the inference path.
|
|
271
|
+
*/
|
|
272
|
+
function appendTurnRecord(record) {
|
|
273
|
+
try {
|
|
274
|
+
if (!fs.existsSync(TELEMETRY_DIR)) {
|
|
275
|
+
fs.mkdirSync(TELEMETRY_DIR, { recursive: true });
|
|
276
|
+
}
|
|
277
|
+
const file = TURNS_LEDGER_FILE();
|
|
278
|
+
if (fs.existsSync(file)) {
|
|
279
|
+
const existing = fs.readFileSync(file, "utf8").split("\n").filter((l) => l.trim() !== "");
|
|
280
|
+
if (existing.length >= MAX_TURNS_LEDGER) {
|
|
281
|
+
const trimmed = existing.slice(existing.length - (MAX_TURNS_LEDGER - 1));
|
|
282
|
+
fs.writeFileSync(file, trimmed.join("\n") + "\n", "utf8");
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
fs.appendFileSync(file, JSON.stringify(record) + "\n", "utf8");
|
|
286
|
+
} catch (err) {
|
|
287
|
+
console.error("[Telemetry] Failed to append turn record:", err.message);
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
/**
|
|
292
|
+
* Records an individual tool execution.
|
|
293
|
+
*/
|
|
294
|
+
export function recordToolExecution({ toolName = "unknown", isError = false } = {}) {
|
|
295
|
+
const stats = getCumulativeTelemetry();
|
|
296
|
+
stats.total_tool_calls = (stats.total_tool_calls || 0) + 1;
|
|
297
|
+
stats.tool_calls[toolName] = (stats.tool_calls[toolName] || 0) + 1;
|
|
298
|
+
|
|
299
|
+
if (isError) {
|
|
300
|
+
stats.total_tool_errors = (stats.total_tool_errors || 0) + 1;
|
|
301
|
+
stats.tool_errors[toolName] = (stats.tool_errors[toolName] || 0) + 1;
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
saveCumulativeTelemetry(stats);
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
/**
|
|
308
|
+
* Records task completion, failure, or cancellation.
|
|
309
|
+
*/
|
|
310
|
+
export function recordTaskResult({
|
|
311
|
+
isSuccess = true,
|
|
312
|
+
isCancelled = false,
|
|
313
|
+
effort = null,
|
|
314
|
+
} = {}) {
|
|
315
|
+
const stats = getCumulativeTelemetry();
|
|
316
|
+
if (isCancelled) {
|
|
317
|
+
stats.total_tasks_cancelled = (stats.total_tasks_cancelled || 0) + 1;
|
|
318
|
+
} else if (isSuccess) {
|
|
319
|
+
stats.total_tasks_completed = (stats.total_tasks_completed || 0) + 1;
|
|
320
|
+
} else {
|
|
321
|
+
stats.total_tasks_failed = (stats.total_tasks_failed || 0) + 1;
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
if (effort && stats.reasoning_effort[effort] !== undefined) {
|
|
325
|
+
stats.reasoning_effort[effort] += 1;
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
saveCumulativeTelemetry(stats);
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
/**
|
|
332
|
+
* Samples live vLLM engine log metrics (prefix cache hit rate, KV usage, DFlash acceptance)
|
|
333
|
+
* without blocking or failing if the engine is stopped or unreachable.
|
|
334
|
+
*/
|
|
335
|
+
export function sampleLiveVllmMetrics() {
|
|
336
|
+
try {
|
|
337
|
+
let logPath = ENGINE_LOG_PATH;
|
|
338
|
+
if (IS_WINDOWS && logPath.startsWith("/")) {
|
|
339
|
+
logPath = `\\\\wsl.localhost\\Ubuntu${logPath.replace(/\//g, "\\")}`;
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
if (!fs.existsSync(logPath)) return;
|
|
343
|
+
|
|
344
|
+
const stats = fs.statSync(logPath);
|
|
345
|
+
const readSize = Math.min(stats.size, 32768);
|
|
346
|
+
const buffer = Buffer.alloc(readSize);
|
|
347
|
+
const fd = fs.openSync(logPath, "r");
|
|
348
|
+
fs.readSync(fd, buffer, 0, readSize, stats.size - readSize);
|
|
349
|
+
fs.closeSync(fd);
|
|
350
|
+
|
|
351
|
+
const tailText = buffer.toString("utf8");
|
|
352
|
+
const telemetry = getCumulativeTelemetry();
|
|
353
|
+
let updated = false;
|
|
354
|
+
|
|
355
|
+
const prefixMatch = tailText.match(/Prefix cache hit rate:\s*([\d\.]+)%/g);
|
|
356
|
+
if (prefixMatch && prefixMatch.length > 0) {
|
|
357
|
+
const last = prefixMatch[prefixMatch.length - 1];
|
|
358
|
+
const val = parseFloat(last.replace(/[^0-9.]/g, ""));
|
|
359
|
+
if (Number.isFinite(val)) {
|
|
360
|
+
telemetry.vllm_engine_metrics.prefix_cache_hit_rate_pct = val;
|
|
361
|
+
updated = true;
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
const kvMatch = tailText.match(/GPU KV cache usage:\s*([\d\.]+)%/g);
|
|
366
|
+
if (kvMatch && kvMatch.length > 0) {
|
|
367
|
+
const last = kvMatch[kvMatch.length - 1];
|
|
368
|
+
const val = parseFloat(last.replace(/[^0-9.]/g, ""));
|
|
369
|
+
if (Number.isFinite(val) && val > telemetry.vllm_engine_metrics.peak_gpu_kv_cache_pct) {
|
|
370
|
+
telemetry.vllm_engine_metrics.peak_gpu_kv_cache_pct = val;
|
|
371
|
+
updated = true;
|
|
372
|
+
}
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
const specMatch = tailText.match(/Mean acceptance length:\s*([\d\.]+)/g);
|
|
376
|
+
if (specMatch && specMatch.length > 0) {
|
|
377
|
+
const last = specMatch[specMatch.length - 1];
|
|
378
|
+
const val = parseFloat(last.replace(/[^0-9.]/g, ""));
|
|
379
|
+
if (Number.isFinite(val)) {
|
|
380
|
+
telemetry.vllm_engine_metrics.spec_mean_acceptance_length = val;
|
|
381
|
+
updated = true;
|
|
382
|
+
}
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
const draftMatch = tailText.match(/Avg Draft acceptance rate:\s*([\d\.]+)%/g);
|
|
386
|
+
if (draftMatch && draftMatch.length > 0) {
|
|
387
|
+
const last = draftMatch[draftMatch.length - 1];
|
|
388
|
+
const val = parseFloat(last.replace(/[^0-9.]/g, ""));
|
|
389
|
+
if (Number.isFinite(val)) {
|
|
390
|
+
telemetry.vllm_engine_metrics.spec_draft_acceptance_rate_pct = val;
|
|
391
|
+
updated = true;
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
if (updated) {
|
|
396
|
+
saveCumulativeTelemetry(telemetry);
|
|
397
|
+
}
|
|
398
|
+
} catch {}
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
/**
|
|
402
|
+
* Reads the rolling per-turn ledger (turns.jsonl) into an array of records.
|
|
403
|
+
* Returns [] when the file is absent or unreadable — a missing ledger is a
|
|
404
|
+
* valid state (fresh install / pre-Slice-2 history) and must not throw.
|
|
405
|
+
*/
|
|
406
|
+
function readTurnsLedger() {
|
|
407
|
+
try {
|
|
408
|
+
const file = TURNS_LEDGER_FILE();
|
|
409
|
+
if (!fs.existsSync(file)) return [];
|
|
410
|
+
return fs
|
|
411
|
+
.readFileSync(file, "utf8")
|
|
412
|
+
.split("\n")
|
|
413
|
+
.filter((l) => l.trim() !== "")
|
|
414
|
+
.map((l) => {
|
|
415
|
+
try {
|
|
416
|
+
return JSON.parse(l);
|
|
417
|
+
} catch {
|
|
418
|
+
return null;
|
|
419
|
+
}
|
|
420
|
+
})
|
|
421
|
+
.filter((r) => r && typeof r === "object");
|
|
422
|
+
} catch {
|
|
423
|
+
return [];
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
/**
|
|
428
|
+
* Resolves a time-slice filter into a [startMs, endMs) window.
|
|
429
|
+
*
|
|
430
|
+
* Supported keys (first match wins, in this precedence order):
|
|
431
|
+
* - since / until : ISO-8601 strings (or ms numbers) bounding the window.
|
|
432
|
+
* - window : "1h" | "24h" | "today" | "yesterday" (relative to now).
|
|
433
|
+
* - date : "YYYY-MM-DD" → that calendar day (local time).
|
|
434
|
+
* - hour : 0-23, combined with `date` (defaults to today).
|
|
435
|
+
* - minute : 0-59, combined with `date`+`hour` (defaults to today).
|
|
436
|
+
*
|
|
437
|
+
* Returns { startMs, endMs } where the window is inclusive [startMs, endMs].
|
|
438
|
+
*/
|
|
439
|
+
function resolveTimeWindow({ since, until, date, hour, minute, window } = {}) {
|
|
440
|
+
const now = Date.now();
|
|
441
|
+
|
|
442
|
+
// Explicit ISO / numeric bounds take precedence.
|
|
443
|
+
if (since !== undefined || until !== undefined) {
|
|
444
|
+
const startMs = since === undefined ? 0 : (typeof since === "number" ? since : Date.parse(since));
|
|
445
|
+
const endMs = until === undefined ? now : (typeof until === "number" ? until : Date.parse(until));
|
|
446
|
+
return {
|
|
447
|
+
startMs: Number.isFinite(startMs) ? startMs : 0,
|
|
448
|
+
endMs: Number.isFinite(endMs) ? endMs : now,
|
|
449
|
+
};
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
// Relative windows.
|
|
453
|
+
if (window === "1h") return { startMs: now - 3600_000, endMs: now };
|
|
454
|
+
if (window === "24h") return { startMs: now - 86400_000, endMs: now };
|
|
455
|
+
if (window === "today") {
|
|
456
|
+
const d = new Date();
|
|
457
|
+
d.setHours(0, 0, 0, 0);
|
|
458
|
+
return { startMs: d.getTime(), endMs: now };
|
|
459
|
+
}
|
|
460
|
+
if (window === "yesterday") {
|
|
461
|
+
const end = new Date();
|
|
462
|
+
end.setHours(0, 0, 0, 0);
|
|
463
|
+
const start = new Date(end);
|
|
464
|
+
start.setDate(start.getDate() - 1);
|
|
465
|
+
return { startMs: start.getTime(), endMs: end.getTime() };
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
// Calendar-date windows (optionally narrowed to an hour or minute).
|
|
469
|
+
if (date !== undefined || hour !== undefined || minute !== undefined) {
|
|
470
|
+
const base = date !== undefined ? new Date(`${date}T00:00:00`) : new Date();
|
|
471
|
+
if (Number.isNaN(base.getTime())) base.setHours(0, 0, 0, 0);
|
|
472
|
+
if (hour !== undefined) base.setHours(hour, 0, 0, 0);
|
|
473
|
+
if (minute !== undefined) base.setMinutes(minute, 0, 0);
|
|
474
|
+
const spanMs = minute !== undefined ? 60_000 : hour !== undefined ? 3600_000 : 86400_000;
|
|
475
|
+
return { startMs: base.getTime(), endMs: base.getTime() + spanMs };
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
// No filter → full lifetime (bounded by the ledger's own extent).
|
|
479
|
+
return { startMs: 0, endMs: now };
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
/**
|
|
483
|
+
* Time-sliced telemetry query.
|
|
484
|
+
*
|
|
485
|
+
* Filters the per-turn ledger to a window (see resolveTimeWindow) and returns
|
|
486
|
+
* aggregate statistics for that slice:
|
|
487
|
+
* {
|
|
488
|
+
* startMs, endMs,
|
|
489
|
+
* turns,
|
|
490
|
+
* promptTokens, completionTokens, reasoningTokens,
|
|
491
|
+
* avg_ttft_ms, avg_generation_ms,
|
|
492
|
+
* avg_decode_tps, avg_prefill_tps, avg_tpot_ms,
|
|
493
|
+
* costSavedUsd,
|
|
494
|
+
* }
|
|
495
|
+
*
|
|
496
|
+
* Averages are computed over the turns that actually reported a finite value
|
|
497
|
+
* for that field (nulls are excluded, not zero-filled). Returns a zeroed
|
|
498
|
+
* aggregate (turns: 0) when no turns match the slice.
|
|
499
|
+
*/
|
|
500
|
+
export function queryTelemetry({ since, until, date, hour, minute, window } = {}) {
|
|
501
|
+
const { startMs, endMs } = resolveTimeWindow({ since, until, date, hour, minute, window });
|
|
502
|
+
const all = readTurnsLedger();
|
|
503
|
+
const turns = all.filter((r) => {
|
|
504
|
+
const t = typeof r.ts === "number" ? r.ts : Date.parse(r.ts);
|
|
505
|
+
// Inclusive [startMs, endMs] window: a turn recorded at exactly the
|
|
506
|
+
// window's end boundary (e.g. "now") is part of the slice, not excluded.
|
|
507
|
+
return Number.isFinite(t) && t >= startMs && t <= endMs;
|
|
508
|
+
});
|
|
509
|
+
|
|
510
|
+
const sum = (key) => turns.reduce((acc, r) => acc + (typeof r[key] === "number" ? r[key] : 0), 0);
|
|
511
|
+
const avg = (key) => {
|
|
512
|
+
const vals = turns.map((r) => r[key]).filter((v) => typeof v === "number" && Number.isFinite(v));
|
|
513
|
+
if (vals.length === 0) return null;
|
|
514
|
+
return Number((vals.reduce((a, b) => a + b, 0) / vals.length).toFixed(2));
|
|
515
|
+
};
|
|
516
|
+
|
|
517
|
+
const promptTokens = sum("prompt");
|
|
518
|
+
const completionTokens = sum("comp");
|
|
519
|
+
|
|
520
|
+
return {
|
|
521
|
+
startMs,
|
|
522
|
+
endMs,
|
|
523
|
+
turns: turns.length,
|
|
524
|
+
promptTokens,
|
|
525
|
+
completionTokens,
|
|
526
|
+
reasoningTokens: sum("reasoning"),
|
|
527
|
+
avg_ttft_ms: avg("ttft"),
|
|
528
|
+
avg_generation_ms: avg("gen"),
|
|
529
|
+
avg_decode_tps: avg("decodeTps"),
|
|
530
|
+
avg_prefill_tps: avg("prefillTps"),
|
|
531
|
+
avg_tpot_ms: avg("tpotMs"),
|
|
532
|
+
costSavedUsd: calculateCostSaved(promptTokens, completionTokens),
|
|
533
|
+
};
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
/**
|
|
537
|
+
* Formats a user-friendly statistics string with rich telemetry.
|
|
538
|
+
*
|
|
539
|
+
* Accepts an optional time-slice filter (same keys as queryTelemetry). When the
|
|
540
|
+
* filter carries at least one defined key, the summary is computed over that
|
|
541
|
+
* slice of the per-turn ledger; otherwise it renders the lifetime cumulative
|
|
542
|
+
* stats.
|
|
543
|
+
*/
|
|
544
|
+
export function formatTelemetrySummary(filter = {}) {
|
|
545
|
+
const hasFilter = Object.keys(filter).some((k) => filter[k] !== undefined);
|
|
546
|
+
|
|
547
|
+
if (hasFilter) {
|
|
548
|
+
let label = filter.window || filter.since || filter.date || "slice";
|
|
549
|
+
if (filter.date && filter.hour !== undefined) {
|
|
550
|
+
label = `${filter.date} ${String(filter.hour).padStart(2, "0")}:00`;
|
|
551
|
+
if (filter.minute !== undefined) {
|
|
552
|
+
label = `${filter.date} ${String(filter.hour).padStart(2, "0")}:${String(filter.minute).padStart(2, "0")}`;
|
|
553
|
+
}
|
|
554
|
+
} else if (filter.hour !== undefined) {
|
|
555
|
+
label = `hour ${filter.hour}`;
|
|
556
|
+
if (filter.minute !== undefined) {
|
|
557
|
+
label = `hour ${filter.hour}:${String(filter.minute).padStart(2, "0")}`;
|
|
558
|
+
}
|
|
559
|
+
}
|
|
560
|
+
const slice = queryTelemetry(filter);
|
|
561
|
+
const fmt = (v, unit, digits = 1) =>
|
|
562
|
+
v === null || v === undefined ? "n/a" : `${Number(v).toFixed(digits)}${unit ? " " + unit : ""}`;
|
|
563
|
+
const lines = [
|
|
564
|
+
`### 📊 Qwen Telemetry — ${label}`,
|
|
565
|
+
`- **Turns in slice**: **${slice.turns.toLocaleString()}**`,
|
|
566
|
+
`- **Tokens**: **${slice.promptTokens.toLocaleString()}** prompt | **${slice.completionTokens.toLocaleString()}** completion | **${slice.reasoningTokens.toLocaleString()}** reasoning`,
|
|
567
|
+
`- **Decode Speed**: **${fmt(slice.avg_decode_tps, "tok/s", 2)}**`,
|
|
568
|
+
`- **Prefill Speed**: **${fmt(slice.avg_prefill_tps, "tok/s", 2)}**`,
|
|
569
|
+
`- **TTFT**: **${slice.avg_ttft_ms != null ? fmt(slice.avg_ttft_ms / 1000, "s", 1) : "n/a"}**`,
|
|
570
|
+
`- **TPOT**: **${fmt(slice.avg_tpot_ms, "ms/tok", 2)}**`,
|
|
571
|
+
`- **Cost Saved (slice)**: **$${slice.costSavedUsd.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 })} USD**`,
|
|
572
|
+
];
|
|
573
|
+
return { summary: lines.join("\n"), stats: slice };
|
|
574
|
+
}
|
|
575
|
+
|
|
576
|
+
const stats = getCumulativeTelemetry();
|
|
577
|
+
const compM = (stats.total_completion_tokens / 1_000_000).toFixed(2);
|
|
578
|
+
const reasoningM = (stats.total_reasoning_tokens / 1_000_000).toFixed(2);
|
|
579
|
+
const promptM = (stats.total_prompt_tokens / 1_000_000).toFixed(2);
|
|
580
|
+
const measuredPromptM = (stats.total_prompt_tokens_measured / 1_000_000).toFixed(2);
|
|
581
|
+
const toolErrorRate = stats.total_tool_calls > 0
|
|
582
|
+
? ((stats.total_tool_errors / stats.total_tool_calls) * 100).toFixed(2)
|
|
583
|
+
: "0.00";
|
|
584
|
+
|
|
585
|
+
const topTools = Object.entries(stats.tool_calls || {})
|
|
586
|
+
.sort((a, b) => b[1] - a[1])
|
|
587
|
+
.slice(0, 5)
|
|
588
|
+
.map(([tool, count]) => `${tool}: ${count.toLocaleString()}`)
|
|
589
|
+
.join(", ");
|
|
590
|
+
|
|
591
|
+
const opusSaved = calculateCostSaved(stats.total_prompt_tokens, stats.total_completion_tokens, 4.0, 20.0);
|
|
592
|
+
const frontierSaved = calculateCostSaved(stats.total_prompt_tokens, stats.total_completion_tokens, 10.0, 50.0);
|
|
593
|
+
const glmSaved = calculateCostSaved(stats.total_prompt_tokens, stats.total_completion_tokens, 1.4, 4.4);
|
|
594
|
+
|
|
595
|
+
const prefillSec = stats.avg_prefill_ms || stats.avg_ttft_ms
|
|
596
|
+
? ((stats.avg_prefill_ms || stats.avg_ttft_ms) / 1000).toFixed(1)
|
|
597
|
+
: null;
|
|
598
|
+
const genSec = stats.avg_generation_ms
|
|
599
|
+
? (stats.avg_generation_ms / 1000).toFixed(1)
|
|
600
|
+
: null;
|
|
601
|
+
const timingItems = [];
|
|
602
|
+
if (prefillSec) timingItems.push(`**${prefillSec}s** avg prefill (TTFT)`);
|
|
603
|
+
if (genSec) timingItems.push(`**${genSec}s** avg generation`);
|
|
604
|
+
const timingLine = timingItems.length > 0
|
|
605
|
+
? `- **Turn Latency**: ${timingItems.join(" | ")}`
|
|
606
|
+
: null;
|
|
607
|
+
|
|
608
|
+
const decodeTps = stats.avg_decode_tps != null ? `**${stats.avg_decode_tps}** tok/s decode` : null;
|
|
609
|
+
const prefillTps = stats.avg_prefill_tps != null ? `**${stats.avg_prefill_tps}** tok/s prefill` : null;
|
|
610
|
+
const tpot = stats.avg_tpot_ms != null ? `**${stats.avg_tpot_ms}** ms/tok TPOT` : null;
|
|
611
|
+
const rateItems = [decodeTps, prefillTps, tpot].filter(Boolean);
|
|
612
|
+
const rateLine = rateItems.length > 0
|
|
613
|
+
? `- **Throughput**: ${rateItems.join(" | ")}`
|
|
614
|
+
: null;
|
|
615
|
+
|
|
616
|
+
const summary = [
|
|
617
|
+
`### 🚀 Lifetime Qwen Usage & Castor Telemetry`,
|
|
618
|
+
`- **Completion Generated**: **${compM}M** tokens (${stats.total_completion_tokens.toLocaleString()} tok)`,
|
|
619
|
+
`- **Deliberative Reasoning**: **${reasoningM}M** thinking tokens (${stats.total_reasoning_tokens.toLocaleString()} tok)`,
|
|
620
|
+
`- **Prompt Prefill**: **${promptM}M** tokens (${measuredPromptM}M exact measured + ${((stats.total_prompt_tokens_estimated || 0) / 1_000_000).toFixed(2)}M estimated)`,
|
|
621
|
+
...(timingLine ? [timingLine] : []),
|
|
622
|
+
...(rateLine ? [rateLine] : []),
|
|
623
|
+
`- **Total Turns & Sessions**: **${stats.total_turns.toLocaleString()}** turns across **${stats.total_sessions}** sessions`,
|
|
624
|
+
`- **Task Lifecycle**: **${stats.total_tasks_completed}** completed, **${stats.total_tasks_failed}** failed, **${stats.total_tasks_cancelled || 0}** cancelled`,
|
|
625
|
+
`- **Tool Execution**: **${stats.total_tool_calls.toLocaleString()}** calls with only **${stats.total_tool_errors}** errors (${toolErrorRate}% error rate)`,
|
|
626
|
+
`- **Top Tools**: ${topTools}`,
|
|
627
|
+
`- **vLLM Acceleration**: **${stats.vllm_engine_metrics.prefix_cache_hit_rate_pct}%** Prefix Cache hit rate | **${stats.vllm_engine_metrics.spec_mean_acceptance_length}** tok/step DFlash2 mean acceptance | **${stats.vllm_engine_metrics.peak_gpu_kv_cache_pct}%** peak GPU KV cache`,
|
|
628
|
+
`- **Financial Value**: **${stats.estimated_cost_saved_usd.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 })} USD** in API cost saved vs **${stats.benchmark_model || BENCHMARK_MODEL}** (${PROMPT_COST_PER_MILLION.toFixed(2)}/M prompt, ${COMPLETION_COST_PER_MILLION.toFixed(2)}/M completion) at **$0 local token cost**`,
|
|
629
|
+
` - *Vs Claude Opus 5.5 ($4/$20)*: **${opusSaved.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 })} USD** saved`,
|
|
630
|
+
` - *Vs GPT-6 Astra / Claude Fable 5.1 ($10/$50)*: **${frontierSaved.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 })} USD** saved`,
|
|
631
|
+
` - *Vs GLM-5.3 ($1.40/$4.40)*: **${glmSaved.toLocaleString("en-US", { minimumFractionDigits: 2, maximumFractionDigits: 2 })} USD** saved`,
|
|
632
|
+
].join("\n");
|
|
633
|
+
|
|
634
|
+
return {
|
|
635
|
+
summary,
|
|
636
|
+
stats,
|
|
637
|
+
};
|
|
638
|
+
}
|