@tokcalc/mcp-server 0.2.0-alpha.2 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/http.js +13836 -7926
- package/dist/index.js +37 -13
- package/dist/server.js +21205 -0
- package/package.json +30 -2
package/dist/index.js
CHANGED
|
@@ -12858,6 +12858,7 @@ function datetime2(params) {
|
|
|
12858
12858
|
}
|
|
12859
12859
|
// node_modules/@modelcontextprotocol/sdk/dist/esm/types.js
|
|
12860
12860
|
var LATEST_PROTOCOL_VERSION = "2025-11-25";
|
|
12861
|
+
var DEFAULT_NEGOTIATED_PROTOCOL_VERSION = "2025-03-26";
|
|
12861
12862
|
var SUPPORTED_PROTOCOL_VERSIONS = [LATEST_PROTOCOL_VERSION, "2025-06-18", "2025-03-26", "2024-11-05", "2024-10-07"];
|
|
12862
12863
|
var RELATED_TASK_META_KEY = "io.modelcontextprotocol/related-task";
|
|
12863
12864
|
var JSONRPC_VERSION = "2.0";
|
|
@@ -13031,6 +13032,7 @@ var InitializeRequestSchema = RequestSchema.extend({
|
|
|
13031
13032
|
method: literal("initialize"),
|
|
13032
13033
|
params: InitializeRequestParamsSchema
|
|
13033
13034
|
});
|
|
13035
|
+
var isInitializeRequest = (value) => InitializeRequestSchema.safeParse(value).success;
|
|
13034
13036
|
var ServerCapabilitiesSchema = object({
|
|
13035
13037
|
experimental: record(string2(), AssertObjectSchema).optional(),
|
|
13036
13038
|
logging: AssertObjectSchema.optional(),
|
|
@@ -20402,7 +20404,7 @@ function calculate(input) {
|
|
|
20402
20404
|
const computeCeiling = prefillTokensPerSec;
|
|
20403
20405
|
const batchCrossover = quant.bytesPerParam * gpuFlops * 1000 * ETA_COMPUTE / (2 * gpu.memBandwidthGbps * ETA_MEM * quantEff * (tp > 1 ? multiGpuEfficiency : 1));
|
|
20404
20406
|
const memoryBoundAggregate = decodeTokensPerSec * input.batchSize * continuousBatchingMultiplier;
|
|
20405
|
-
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) : 0;
|
|
20407
|
+
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) * tp : 0;
|
|
20406
20408
|
const cachePrefixTokens = input.cachePrefixTokens ?? 0;
|
|
20407
20409
|
const cacheHitRate = input.cacheHitRate ?? 0;
|
|
20408
20410
|
const cacheTTL = input.cacheTTL ?? "none";
|
|
@@ -20866,7 +20868,8 @@ var CompareGpusSchema = object({
|
|
|
20866
20868
|
promptTokens: number2().int().min(1).max(2000000).default(500),
|
|
20867
20869
|
outputTokens: number2().int().min(1).max(200000).default(200),
|
|
20868
20870
|
sortBy: _enum(["lowest_cost", "highest_throughput", "best_value"]).default("best_value"),
|
|
20869
|
-
limit: number2().int().min(1).max(30).default(10)
|
|
20871
|
+
limit: number2().int().min(1).max(30).default(10),
|
|
20872
|
+
gpus: array(string2()).optional().describe("Restrict comparison to specific GPU IDs (e.g. ['h100-sxm','h200-sxm']). Use list_gpus first to find IDs.")
|
|
20870
20873
|
});
|
|
20871
20874
|
var RecommendTopologySchema = object({
|
|
20872
20875
|
model: ModelIdSchema,
|
|
@@ -20971,7 +20974,25 @@ function handleCompareGpus(input) {
|
|
|
20971
20974
|
const quant = QUANT_MAP[input.quantization];
|
|
20972
20975
|
if (!model || !quant)
|
|
20973
20976
|
return { error: "Unknown model or quantization", models: Object.keys(MODEL_MAP).slice(0, 10) };
|
|
20974
|
-
|
|
20977
|
+
let candidatePool = GPUS;
|
|
20978
|
+
if (input.gpus && input.gpus.length > 0) {
|
|
20979
|
+
const requested = new Set(input.gpus);
|
|
20980
|
+
candidatePool = GPUS.filter((g) => requested.has(g.id));
|
|
20981
|
+
if (candidatePool.length === 0) {
|
|
20982
|
+
return {
|
|
20983
|
+
error: `No GPUs matched gpus=[${input.gpus.join(", ")}]. Call list_gpus to find valid IDs.`,
|
|
20984
|
+
validGpuIds: GPUS.slice(0, 10).map((g) => g.id)
|
|
20985
|
+
};
|
|
20986
|
+
}
|
|
20987
|
+
}
|
|
20988
|
+
const skippedNoPrice = [];
|
|
20989
|
+
const candidates = candidatePool.filter((g) => {
|
|
20990
|
+
const fitsMemory = g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam;
|
|
20991
|
+
const hasPrice = g.usdPerHour !== null && g.usdPerHour !== undefined && g.usdPerHour > 0;
|
|
20992
|
+
if (!hasPrice)
|
|
20993
|
+
skippedNoPrice.push(g.id);
|
|
20994
|
+
return fitsMemory && hasPrice;
|
|
20995
|
+
});
|
|
20975
20996
|
const results = candidates.map((g) => {
|
|
20976
20997
|
const r = calculate({
|
|
20977
20998
|
modelId: input.model,
|
|
@@ -21004,10 +21025,11 @@ function handleCompareGpus(input) {
|
|
|
21004
21025
|
return b.aggregateTokensPerSecond / Math.max(b.costPerMillionTokens, 0.001) - a.aggregateTokensPerSecond / Math.max(a.costPerMillionTokens, 0.001);
|
|
21005
21026
|
}).slice(0, input.limit);
|
|
21006
21027
|
return {
|
|
21007
|
-
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens}/M tokens.`,
|
|
21028
|
+
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens.toLocaleString("en-US")}/M tokens.`,
|
|
21008
21029
|
comparisons: sorted,
|
|
21009
21030
|
totalCandidates: results.length,
|
|
21010
21031
|
sortBy: input.sortBy,
|
|
21032
|
+
skippedNoPrice: skippedNoPrice.length > 0 ? skippedNoPrice : undefined,
|
|
21011
21033
|
assumptions: [`Region: us-east-1 (default)`, `On-demand pricing`, `η_mem = 0.65`, `Catalog version: 0.3.0`]
|
|
21012
21034
|
};
|
|
21013
21035
|
}
|
|
@@ -21016,10 +21038,10 @@ function handleRecommendTopology(input) {
|
|
|
21016
21038
|
const quant = QUANT_MAP[input.quantization];
|
|
21017
21039
|
if (!model || !quant)
|
|
21018
21040
|
return { error: "Unknown model or quantization" };
|
|
21041
|
+
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, input.batchSize);
|
|
21019
21042
|
const results = GPUS.map((g) => {
|
|
21020
21043
|
const rec = recommendTopology(model, g, input.contextTokens, input.batchSize, quant.bytesPerParam);
|
|
21021
|
-
const maxConcurrent = computeMaxConcurrency(model, g,
|
|
21022
|
-
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, 1);
|
|
21044
|
+
const maxConcurrent = computeMaxConcurrency(model, g, input.batchSize, input.contextTokens, quant.bytesPerParam);
|
|
21023
21045
|
return {
|
|
21024
21046
|
gpuId: g.id,
|
|
21025
21047
|
gpuName: g.name,
|
|
@@ -21033,12 +21055,12 @@ function handleRecommendTopology(input) {
|
|
|
21033
21055
|
};
|
|
21034
21056
|
}).filter((r) => r.fits).slice(0, 5);
|
|
21035
21057
|
return {
|
|
21036
|
-
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context: ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
21058
|
+
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context (batch=${input.batchSize}): ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
21037
21059
|
model: { name: model.name, paramsB: model.paramsB, activeParamsB: model.activeParamsB, isMoE: model.isMoE },
|
|
21038
|
-
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens) },
|
|
21060
|
+
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens), batchSize: input.batchSize },
|
|
21039
21061
|
recommendations: results,
|
|
21040
|
-
formula: `KV per
|
|
21041
|
-
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
21062
|
+
formula: `KV per batch = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens × ${input.batchSize} batch = ${kvPerRequest.toFixed(2)} GB`,
|
|
21063
|
+
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV × batchSize must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
21042
21064
|
};
|
|
21043
21065
|
}
|
|
21044
21066
|
function handleEstimateApiVsSelfHost(input) {
|
|
@@ -21120,8 +21142,10 @@ function handleListGpus(input) {
|
|
|
21120
21142
|
filtered = filtered.filter((g) => g.vendor === input.vendor);
|
|
21121
21143
|
if (input.category)
|
|
21122
21144
|
filtered = filtered.filter((g) => g.category === input.category);
|
|
21123
|
-
if (input.minVramGb !== undefined)
|
|
21124
|
-
|
|
21145
|
+
if (input.minVramGb !== undefined) {
|
|
21146
|
+
const minVram = input.minVramGb;
|
|
21147
|
+
filtered = filtered.filter((g) => g.vramGb >= minVram);
|
|
21148
|
+
}
|
|
21125
21149
|
return {
|
|
21126
21150
|
count: filtered.length,
|
|
21127
21151
|
gpus: filtered.map((g) => ({
|
|
@@ -21212,7 +21236,7 @@ var TOOL_DEFINITIONS = [
|
|
|
21212
21236
|
}
|
|
21213
21237
|
];
|
|
21214
21238
|
function createMcpServer() {
|
|
21215
|
-
const server = new Server({ name: "tokcalc", version: "0.2.0
|
|
21239
|
+
const server = new Server({ name: "tokcalc", version: "0.2.0" }, {
|
|
21216
21240
|
capabilities: {
|
|
21217
21241
|
tools: {}
|
|
21218
21242
|
}
|