@tokcalc/mcp-server 0.2.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (4) hide show
  1. package/dist/http.js +30 -10
  2. package/dist/index.js +30 -10
  3. package/dist/server.js +3662 -8484
  4. package/package.json +64 -64
package/dist/http.js CHANGED
@@ -24163,7 +24163,7 @@ function calculate(input) {
24163
24163
  const computeCeiling = prefillTokensPerSec;
24164
24164
  const batchCrossover = quant.bytesPerParam * gpuFlops * 1000 * ETA_COMPUTE / (2 * gpu.memBandwidthGbps * ETA_MEM * quantEff * (tp > 1 ? multiGpuEfficiency : 1));
24165
24165
  const memoryBoundAggregate = decodeTokensPerSec * input.batchSize * continuousBatchingMultiplier;
24166
- const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) : 0;
24166
+ const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) * tp : 0;
24167
24167
  const cachePrefixTokens = input.cachePrefixTokens ?? 0;
24168
24168
  const cacheHitRate = input.cacheHitRate ?? 0;
24169
24169
  const cacheTTL = input.cacheTTL ?? "none";
@@ -24627,7 +24627,8 @@ var CompareGpusSchema = object({
24627
24627
  promptTokens: number2().int().min(1).max(2000000).default(500),
24628
24628
  outputTokens: number2().int().min(1).max(200000).default(200),
24629
24629
  sortBy: _enum(["lowest_cost", "highest_throughput", "best_value"]).default("best_value"),
24630
- limit: number2().int().min(1).max(30).default(10)
24630
+ limit: number2().int().min(1).max(30).default(10),
24631
+ gpus: array(string2()).optional().describe("Restrict comparison to specific GPU IDs (e.g. ['h100-sxm','h200-sxm']). Use list_gpus first to find IDs.")
24631
24632
  });
24632
24633
  var RecommendTopologySchema = object({
24633
24634
  model: ModelIdSchema,
@@ -24732,7 +24733,25 @@ function handleCompareGpus(input) {
24732
24733
  const quant = QUANT_MAP[input.quantization];
24733
24734
  if (!model || !quant)
24734
24735
  return { error: "Unknown model or quantization", models: Object.keys(MODEL_MAP).slice(0, 10) };
24735
- const candidates = GPUS.filter((g) => g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam);
24736
+ let candidatePool = GPUS;
24737
+ if (input.gpus && input.gpus.length > 0) {
24738
+ const requested = new Set(input.gpus);
24739
+ candidatePool = GPUS.filter((g) => requested.has(g.id));
24740
+ if (candidatePool.length === 0) {
24741
+ return {
24742
+ error: `No GPUs matched gpus=[${input.gpus.join(", ")}]. Call list_gpus to find valid IDs.`,
24743
+ validGpuIds: GPUS.slice(0, 10).map((g) => g.id)
24744
+ };
24745
+ }
24746
+ }
24747
+ const skippedNoPrice = [];
24748
+ const candidates = candidatePool.filter((g) => {
24749
+ const fitsMemory = g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam;
24750
+ const hasPrice = g.usdPerHour !== null && g.usdPerHour !== undefined && g.usdPerHour > 0;
24751
+ if (!hasPrice)
24752
+ skippedNoPrice.push(g.id);
24753
+ return fitsMemory && hasPrice;
24754
+ });
24736
24755
  const results = candidates.map((g) => {
24737
24756
  const r = calculate({
24738
24757
  modelId: input.model,
@@ -24765,10 +24784,11 @@ function handleCompareGpus(input) {
24765
24784
  return b.aggregateTokensPerSecond / Math.max(b.costPerMillionTokens, 0.001) - a.aggregateTokensPerSecond / Math.max(a.costPerMillionTokens, 0.001);
24766
24785
  }).slice(0, input.limit);
24767
24786
  return {
24768
- summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens}/M tokens.`,
24787
+ summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens.toLocaleString("en-US")}/M tokens.`,
24769
24788
  comparisons: sorted,
24770
24789
  totalCandidates: results.length,
24771
24790
  sortBy: input.sortBy,
24791
+ skippedNoPrice: skippedNoPrice.length > 0 ? skippedNoPrice : undefined,
24772
24792
  assumptions: [`Region: us-east-1 (default)`, `On-demand pricing`, `η_mem = 0.65`, `Catalog version: 0.3.0`]
24773
24793
  };
24774
24794
  }
@@ -24777,10 +24797,10 @@ function handleRecommendTopology(input) {
24777
24797
  const quant = QUANT_MAP[input.quantization];
24778
24798
  if (!model || !quant)
24779
24799
  return { error: "Unknown model or quantization" };
24800
+ const kvPerRequest = computeKVCacheGb(model, input.contextTokens, input.batchSize);
24780
24801
  const results = GPUS.map((g) => {
24781
24802
  const rec = recommendTopology(model, g, input.contextTokens, input.batchSize, quant.bytesPerParam);
24782
- const maxConcurrent = computeMaxConcurrency(model, g, 1, input.contextTokens, quant.bytesPerParam);
24783
- const kvPerRequest = computeKVCacheGb(model, input.contextTokens, 1);
24803
+ const maxConcurrent = computeMaxConcurrency(model, g, input.batchSize, input.contextTokens, quant.bytesPerParam);
24784
24804
  return {
24785
24805
  gpuId: g.id,
24786
24806
  gpuName: g.name,
@@ -24794,12 +24814,12 @@ function handleRecommendTopology(input) {
24794
24814
  };
24795
24815
  }).filter((r) => r.fits).slice(0, 5);
24796
24816
  return {
24797
- summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context: ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
24817
+ summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context (batch=${input.batchSize}): ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
24798
24818
  model: { name: model.name, paramsB: model.paramsB, activeParamsB: model.activeParamsB, isMoE: model.isMoE },
24799
- context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens) },
24819
+ context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens), batchSize: input.batchSize },
24800
24820
  recommendations: results,
24801
- formula: `KV per request = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens = ${computeKVCacheGb(model, input.contextTokens, 1).toFixed(2)} GB`,
24802
- assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV must fit in total VRAM`, `Catalog version: 0.3.0`]
24821
+ formula: `KV per batch = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens × ${input.batchSize} batch = ${kvPerRequest.toFixed(2)} GB`,
24822
+ assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV × batchSize must fit in total VRAM`, `Catalog version: 0.3.0`]
24803
24823
  };
24804
24824
  }
24805
24825
  function handleEstimateApiVsSelfHost(input) {
package/dist/index.js CHANGED
@@ -20404,7 +20404,7 @@ function calculate(input) {
20404
20404
  const computeCeiling = prefillTokensPerSec;
20405
20405
  const batchCrossover = quant.bytesPerParam * gpuFlops * 1000 * ETA_COMPUTE / (2 * gpu.memBandwidthGbps * ETA_MEM * quantEff * (tp > 1 ? multiGpuEfficiency : 1));
20406
20406
  const memoryBoundAggregate = decodeTokensPerSec * input.batchSize * continuousBatchingMultiplier;
20407
- const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) : 0;
20407
+ const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) * tp : 0;
20408
20408
  const cachePrefixTokens = input.cachePrefixTokens ?? 0;
20409
20409
  const cacheHitRate = input.cacheHitRate ?? 0;
20410
20410
  const cacheTTL = input.cacheTTL ?? "none";
@@ -20868,7 +20868,8 @@ var CompareGpusSchema = object({
20868
20868
  promptTokens: number2().int().min(1).max(2000000).default(500),
20869
20869
  outputTokens: number2().int().min(1).max(200000).default(200),
20870
20870
  sortBy: _enum(["lowest_cost", "highest_throughput", "best_value"]).default("best_value"),
20871
- limit: number2().int().min(1).max(30).default(10)
20871
+ limit: number2().int().min(1).max(30).default(10),
20872
+ gpus: array(string2()).optional().describe("Restrict comparison to specific GPU IDs (e.g. ['h100-sxm','h200-sxm']). Use list_gpus first to find IDs.")
20872
20873
  });
20873
20874
  var RecommendTopologySchema = object({
20874
20875
  model: ModelIdSchema,
@@ -20973,7 +20974,25 @@ function handleCompareGpus(input) {
20973
20974
  const quant = QUANT_MAP[input.quantization];
20974
20975
  if (!model || !quant)
20975
20976
  return { error: "Unknown model or quantization", models: Object.keys(MODEL_MAP).slice(0, 10) };
20976
- const candidates = GPUS.filter((g) => g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam);
20977
+ let candidatePool = GPUS;
20978
+ if (input.gpus && input.gpus.length > 0) {
20979
+ const requested = new Set(input.gpus);
20980
+ candidatePool = GPUS.filter((g) => requested.has(g.id));
20981
+ if (candidatePool.length === 0) {
20982
+ return {
20983
+ error: `No GPUs matched gpus=[${input.gpus.join(", ")}]. Call list_gpus to find valid IDs.`,
20984
+ validGpuIds: GPUS.slice(0, 10).map((g) => g.id)
20985
+ };
20986
+ }
20987
+ }
20988
+ const skippedNoPrice = [];
20989
+ const candidates = candidatePool.filter((g) => {
20990
+ const fitsMemory = g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam;
20991
+ const hasPrice = g.usdPerHour !== null && g.usdPerHour !== undefined && g.usdPerHour > 0;
20992
+ if (!hasPrice)
20993
+ skippedNoPrice.push(g.id);
20994
+ return fitsMemory && hasPrice;
20995
+ });
20977
20996
  const results = candidates.map((g) => {
20978
20997
  const r = calculate({
20979
20998
  modelId: input.model,
@@ -21006,10 +21025,11 @@ function handleCompareGpus(input) {
21006
21025
  return b.aggregateTokensPerSecond / Math.max(b.costPerMillionTokens, 0.001) - a.aggregateTokensPerSecond / Math.max(a.costPerMillionTokens, 0.001);
21007
21026
  }).slice(0, input.limit);
21008
21027
  return {
21009
- summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens}/M tokens.`,
21028
+ summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens.toLocaleString("en-US")}/M tokens.`,
21010
21029
  comparisons: sorted,
21011
21030
  totalCandidates: results.length,
21012
21031
  sortBy: input.sortBy,
21032
+ skippedNoPrice: skippedNoPrice.length > 0 ? skippedNoPrice : undefined,
21013
21033
  assumptions: [`Region: us-east-1 (default)`, `On-demand pricing`, `η_mem = 0.65`, `Catalog version: 0.3.0`]
21014
21034
  };
21015
21035
  }
@@ -21018,10 +21038,10 @@ function handleRecommendTopology(input) {
21018
21038
  const quant = QUANT_MAP[input.quantization];
21019
21039
  if (!model || !quant)
21020
21040
  return { error: "Unknown model or quantization" };
21041
+ const kvPerRequest = computeKVCacheGb(model, input.contextTokens, input.batchSize);
21021
21042
  const results = GPUS.map((g) => {
21022
21043
  const rec = recommendTopology(model, g, input.contextTokens, input.batchSize, quant.bytesPerParam);
21023
- const maxConcurrent = computeMaxConcurrency(model, g, 1, input.contextTokens, quant.bytesPerParam);
21024
- const kvPerRequest = computeKVCacheGb(model, input.contextTokens, 1);
21044
+ const maxConcurrent = computeMaxConcurrency(model, g, input.batchSize, input.contextTokens, quant.bytesPerParam);
21025
21045
  return {
21026
21046
  gpuId: g.id,
21027
21047
  gpuName: g.name,
@@ -21035,12 +21055,12 @@ function handleRecommendTopology(input) {
21035
21055
  };
21036
21056
  }).filter((r) => r.fits).slice(0, 5);
21037
21057
  return {
21038
- summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context: ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
21058
+ summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context (batch=${input.batchSize}): ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
21039
21059
  model: { name: model.name, paramsB: model.paramsB, activeParamsB: model.activeParamsB, isMoE: model.isMoE },
21040
- context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens) },
21060
+ context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens), batchSize: input.batchSize },
21041
21061
  recommendations: results,
21042
- formula: `KV per request = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens = ${computeKVCacheGb(model, input.contextTokens, 1).toFixed(2)} GB`,
21043
- assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV must fit in total VRAM`, `Catalog version: 0.3.0`]
21062
+ formula: `KV per batch = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens × ${input.batchSize} batch = ${kvPerRequest.toFixed(2)} GB`,
21063
+ assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV × batchSize must fit in total VRAM`, `Catalog version: 0.3.0`]
21044
21064
  };
21045
21065
  }
21046
21066
  function handleEstimateApiVsSelfHost(input) {