@tokcalc/mcp-server 0.2.0-alpha.2 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (4) hide show
  1. package/dist/http.js +13836 -7926
  2. package/dist/index.js +37 -13
  3. package/dist/server.js +21205 -0
  4. package/package.json +30 -2
package/dist/index.js CHANGED
@@ -12858,6 +12858,7 @@ function datetime2(params) {
12858
12858
  }
12859
12859
  // node_modules/@modelcontextprotocol/sdk/dist/esm/types.js
12860
12860
  var LATEST_PROTOCOL_VERSION = "2025-11-25";
12861
+ var DEFAULT_NEGOTIATED_PROTOCOL_VERSION = "2025-03-26";
12861
12862
  var SUPPORTED_PROTOCOL_VERSIONS = [LATEST_PROTOCOL_VERSION, "2025-06-18", "2025-03-26", "2024-11-05", "2024-10-07"];
12862
12863
  var RELATED_TASK_META_KEY = "io.modelcontextprotocol/related-task";
12863
12864
  var JSONRPC_VERSION = "2.0";
@@ -13031,6 +13032,7 @@ var InitializeRequestSchema = RequestSchema.extend({
13031
13032
  method: literal("initialize"),
13032
13033
  params: InitializeRequestParamsSchema
13033
13034
  });
13035
+ var isInitializeRequest = (value) => InitializeRequestSchema.safeParse(value).success;
13034
13036
  var ServerCapabilitiesSchema = object({
13035
13037
  experimental: record(string2(), AssertObjectSchema).optional(),
13036
13038
  logging: AssertObjectSchema.optional(),
@@ -20402,7 +20404,7 @@ function calculate(input) {
20402
20404
  const computeCeiling = prefillTokensPerSec;
20403
20405
  const batchCrossover = quant.bytesPerParam * gpuFlops * 1000 * ETA_COMPUTE / (2 * gpu.memBandwidthGbps * ETA_MEM * quantEff * (tp > 1 ? multiGpuEfficiency : 1));
20404
20406
  const memoryBoundAggregate = decodeTokensPerSec * input.batchSize * continuousBatchingMultiplier;
20405
- const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) : 0;
20407
+ const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) * tp : 0;
20406
20408
  const cachePrefixTokens = input.cachePrefixTokens ?? 0;
20407
20409
  const cacheHitRate = input.cacheHitRate ?? 0;
20408
20410
  const cacheTTL = input.cacheTTL ?? "none";
@@ -20866,7 +20868,8 @@ var CompareGpusSchema = object({
20866
20868
  promptTokens: number2().int().min(1).max(2000000).default(500),
20867
20869
  outputTokens: number2().int().min(1).max(200000).default(200),
20868
20870
  sortBy: _enum(["lowest_cost", "highest_throughput", "best_value"]).default("best_value"),
20869
- limit: number2().int().min(1).max(30).default(10)
20871
+ limit: number2().int().min(1).max(30).default(10),
20872
+ gpus: array(string2()).optional().describe("Restrict comparison to specific GPU IDs (e.g. ['h100-sxm','h200-sxm']). Use list_gpus first to find IDs.")
20870
20873
  });
20871
20874
  var RecommendTopologySchema = object({
20872
20875
  model: ModelIdSchema,
@@ -20971,7 +20974,25 @@ function handleCompareGpus(input) {
20971
20974
  const quant = QUANT_MAP[input.quantization];
20972
20975
  if (!model || !quant)
20973
20976
  return { error: "Unknown model or quantization", models: Object.keys(MODEL_MAP).slice(0, 10) };
20974
- const candidates = GPUS.filter((g) => g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam);
20977
+ let candidatePool = GPUS;
20978
+ if (input.gpus && input.gpus.length > 0) {
20979
+ const requested = new Set(input.gpus);
20980
+ candidatePool = GPUS.filter((g) => requested.has(g.id));
20981
+ if (candidatePool.length === 0) {
20982
+ return {
20983
+ error: `No GPUs matched gpus=[${input.gpus.join(", ")}]. Call list_gpus to find valid IDs.`,
20984
+ validGpuIds: GPUS.slice(0, 10).map((g) => g.id)
20985
+ };
20986
+ }
20987
+ }
20988
+ const skippedNoPrice = [];
20989
+ const candidates = candidatePool.filter((g) => {
20990
+ const fitsMemory = g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam;
20991
+ const hasPrice = g.usdPerHour !== null && g.usdPerHour !== undefined && g.usdPerHour > 0;
20992
+ if (!hasPrice)
20993
+ skippedNoPrice.push(g.id);
20994
+ return fitsMemory && hasPrice;
20995
+ });
20975
20996
  const results = candidates.map((g) => {
20976
20997
  const r = calculate({
20977
20998
  modelId: input.model,
@@ -21004,10 +21025,11 @@ function handleCompareGpus(input) {
21004
21025
  return b.aggregateTokensPerSecond / Math.max(b.costPerMillionTokens, 0.001) - a.aggregateTokensPerSecond / Math.max(a.costPerMillionTokens, 0.001);
21005
21026
  }).slice(0, input.limit);
21006
21027
  return {
21007
- summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens}/M tokens.`,
21028
+ summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens.toLocaleString("en-US")}/M tokens.`,
21008
21029
  comparisons: sorted,
21009
21030
  totalCandidates: results.length,
21010
21031
  sortBy: input.sortBy,
21032
+ skippedNoPrice: skippedNoPrice.length > 0 ? skippedNoPrice : undefined,
21011
21033
  assumptions: [`Region: us-east-1 (default)`, `On-demand pricing`, `η_mem = 0.65`, `Catalog version: 0.3.0`]
21012
21034
  };
21013
21035
  }
@@ -21016,10 +21038,10 @@ function handleRecommendTopology(input) {
21016
21038
  const quant = QUANT_MAP[input.quantization];
21017
21039
  if (!model || !quant)
21018
21040
  return { error: "Unknown model or quantization" };
21041
+ const kvPerRequest = computeKVCacheGb(model, input.contextTokens, input.batchSize);
21019
21042
  const results = GPUS.map((g) => {
21020
21043
  const rec = recommendTopology(model, g, input.contextTokens, input.batchSize, quant.bytesPerParam);
21021
- const maxConcurrent = computeMaxConcurrency(model, g, 1, input.contextTokens, quant.bytesPerParam);
21022
- const kvPerRequest = computeKVCacheGb(model, input.contextTokens, 1);
21044
+ const maxConcurrent = computeMaxConcurrency(model, g, input.batchSize, input.contextTokens, quant.bytesPerParam);
21023
21045
  return {
21024
21046
  gpuId: g.id,
21025
21047
  gpuName: g.name,
@@ -21033,12 +21055,12 @@ function handleRecommendTopology(input) {
21033
21055
  };
21034
21056
  }).filter((r) => r.fits).slice(0, 5);
21035
21057
  return {
21036
- summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context: ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
21058
+ summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context (batch=${input.batchSize}): ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
21037
21059
  model: { name: model.name, paramsB: model.paramsB, activeParamsB: model.activeParamsB, isMoE: model.isMoE },
21038
- context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens) },
21060
+ context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens), batchSize: input.batchSize },
21039
21061
  recommendations: results,
21040
- formula: `KV per request = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens = ${computeKVCacheGb(model, input.contextTokens, 1).toFixed(2)} GB`,
21041
- assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV must fit in total VRAM`, `Catalog version: 0.3.0`]
21062
+ formula: `KV per batch = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens × ${input.batchSize} batch = ${kvPerRequest.toFixed(2)} GB`,
21063
+ assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV × batchSize must fit in total VRAM`, `Catalog version: 0.3.0`]
21042
21064
  };
21043
21065
  }
21044
21066
  function handleEstimateApiVsSelfHost(input) {
@@ -21120,8 +21142,10 @@ function handleListGpus(input) {
21120
21142
  filtered = filtered.filter((g) => g.vendor === input.vendor);
21121
21143
  if (input.category)
21122
21144
  filtered = filtered.filter((g) => g.category === input.category);
21123
- if (input.minVramGb !== undefined)
21124
- filtered = filtered.filter((g) => g.vramGb >= input.minVramGb);
21145
+ if (input.minVramGb !== undefined) {
21146
+ const minVram = input.minVramGb;
21147
+ filtered = filtered.filter((g) => g.vramGb >= minVram);
21148
+ }
21125
21149
  return {
21126
21150
  count: filtered.length,
21127
21151
  gpus: filtered.map((g) => ({
@@ -21212,7 +21236,7 @@ var TOOL_DEFINITIONS = [
21212
21236
  }
21213
21237
  ];
21214
21238
  function createMcpServer() {
21215
- const server = new Server({ name: "tokcalc", version: "0.2.0-alpha.1" }, {
21239
+ const server = new Server({ name: "tokcalc", version: "0.2.0" }, {
21216
21240
  capabilities: {
21217
21241
  tools: {}
21218
21242
  }