@tokcalc/mcp-server 0.2.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/http.js +30 -10
- package/dist/index.js +30 -10
- package/dist/server.js +3662 -8484
- package/package.json +64 -64
package/dist/http.js
CHANGED
|
@@ -24163,7 +24163,7 @@ function calculate(input) {
|
|
|
24163
24163
|
const computeCeiling = prefillTokensPerSec;
|
|
24164
24164
|
const batchCrossover = quant.bytesPerParam * gpuFlops * 1000 * ETA_COMPUTE / (2 * gpu.memBandwidthGbps * ETA_MEM * quantEff * (tp > 1 ? multiGpuEfficiency : 1));
|
|
24165
24165
|
const memoryBoundAggregate = decodeTokensPerSec * input.batchSize * continuousBatchingMultiplier;
|
|
24166
|
-
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) : 0;
|
|
24166
|
+
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) * tp : 0;
|
|
24167
24167
|
const cachePrefixTokens = input.cachePrefixTokens ?? 0;
|
|
24168
24168
|
const cacheHitRate = input.cacheHitRate ?? 0;
|
|
24169
24169
|
const cacheTTL = input.cacheTTL ?? "none";
|
|
@@ -24627,7 +24627,8 @@ var CompareGpusSchema = object({
|
|
|
24627
24627
|
promptTokens: number2().int().min(1).max(2000000).default(500),
|
|
24628
24628
|
outputTokens: number2().int().min(1).max(200000).default(200),
|
|
24629
24629
|
sortBy: _enum(["lowest_cost", "highest_throughput", "best_value"]).default("best_value"),
|
|
24630
|
-
limit: number2().int().min(1).max(30).default(10)
|
|
24630
|
+
limit: number2().int().min(1).max(30).default(10),
|
|
24631
|
+
gpus: array(string2()).optional().describe("Restrict comparison to specific GPU IDs (e.g. ['h100-sxm','h200-sxm']). Use list_gpus first to find IDs.")
|
|
24631
24632
|
});
|
|
24632
24633
|
var RecommendTopologySchema = object({
|
|
24633
24634
|
model: ModelIdSchema,
|
|
@@ -24732,7 +24733,25 @@ function handleCompareGpus(input) {
|
|
|
24732
24733
|
const quant = QUANT_MAP[input.quantization];
|
|
24733
24734
|
if (!model || !quant)
|
|
24734
24735
|
return { error: "Unknown model or quantization", models: Object.keys(MODEL_MAP).slice(0, 10) };
|
|
24735
|
-
|
|
24736
|
+
let candidatePool = GPUS;
|
|
24737
|
+
if (input.gpus && input.gpus.length > 0) {
|
|
24738
|
+
const requested = new Set(input.gpus);
|
|
24739
|
+
candidatePool = GPUS.filter((g) => requested.has(g.id));
|
|
24740
|
+
if (candidatePool.length === 0) {
|
|
24741
|
+
return {
|
|
24742
|
+
error: `No GPUs matched gpus=[${input.gpus.join(", ")}]. Call list_gpus to find valid IDs.`,
|
|
24743
|
+
validGpuIds: GPUS.slice(0, 10).map((g) => g.id)
|
|
24744
|
+
};
|
|
24745
|
+
}
|
|
24746
|
+
}
|
|
24747
|
+
const skippedNoPrice = [];
|
|
24748
|
+
const candidates = candidatePool.filter((g) => {
|
|
24749
|
+
const fitsMemory = g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam;
|
|
24750
|
+
const hasPrice = g.usdPerHour !== null && g.usdPerHour !== undefined && g.usdPerHour > 0;
|
|
24751
|
+
if (!hasPrice)
|
|
24752
|
+
skippedNoPrice.push(g.id);
|
|
24753
|
+
return fitsMemory && hasPrice;
|
|
24754
|
+
});
|
|
24736
24755
|
const results = candidates.map((g) => {
|
|
24737
24756
|
const r = calculate({
|
|
24738
24757
|
modelId: input.model,
|
|
@@ -24765,10 +24784,11 @@ function handleCompareGpus(input) {
|
|
|
24765
24784
|
return b.aggregateTokensPerSecond / Math.max(b.costPerMillionTokens, 0.001) - a.aggregateTokensPerSecond / Math.max(a.costPerMillionTokens, 0.001);
|
|
24766
24785
|
}).slice(0, input.limit);
|
|
24767
24786
|
return {
|
|
24768
|
-
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens}/M tokens.`,
|
|
24787
|
+
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens.toLocaleString("en-US")}/M tokens.`,
|
|
24769
24788
|
comparisons: sorted,
|
|
24770
24789
|
totalCandidates: results.length,
|
|
24771
24790
|
sortBy: input.sortBy,
|
|
24791
|
+
skippedNoPrice: skippedNoPrice.length > 0 ? skippedNoPrice : undefined,
|
|
24772
24792
|
assumptions: [`Region: us-east-1 (default)`, `On-demand pricing`, `η_mem = 0.65`, `Catalog version: 0.3.0`]
|
|
24773
24793
|
};
|
|
24774
24794
|
}
|
|
@@ -24777,10 +24797,10 @@ function handleRecommendTopology(input) {
|
|
|
24777
24797
|
const quant = QUANT_MAP[input.quantization];
|
|
24778
24798
|
if (!model || !quant)
|
|
24779
24799
|
return { error: "Unknown model or quantization" };
|
|
24800
|
+
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, input.batchSize);
|
|
24780
24801
|
const results = GPUS.map((g) => {
|
|
24781
24802
|
const rec = recommendTopology(model, g, input.contextTokens, input.batchSize, quant.bytesPerParam);
|
|
24782
|
-
const maxConcurrent = computeMaxConcurrency(model, g,
|
|
24783
|
-
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, 1);
|
|
24803
|
+
const maxConcurrent = computeMaxConcurrency(model, g, input.batchSize, input.contextTokens, quant.bytesPerParam);
|
|
24784
24804
|
return {
|
|
24785
24805
|
gpuId: g.id,
|
|
24786
24806
|
gpuName: g.name,
|
|
@@ -24794,12 +24814,12 @@ function handleRecommendTopology(input) {
|
|
|
24794
24814
|
};
|
|
24795
24815
|
}).filter((r) => r.fits).slice(0, 5);
|
|
24796
24816
|
return {
|
|
24797
|
-
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context: ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
24817
|
+
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context (batch=${input.batchSize}): ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
24798
24818
|
model: { name: model.name, paramsB: model.paramsB, activeParamsB: model.activeParamsB, isMoE: model.isMoE },
|
|
24799
|
-
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens) },
|
|
24819
|
+
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens), batchSize: input.batchSize },
|
|
24800
24820
|
recommendations: results,
|
|
24801
|
-
formula: `KV per
|
|
24802
|
-
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
24821
|
+
formula: `KV per batch = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens × ${input.batchSize} batch = ${kvPerRequest.toFixed(2)} GB`,
|
|
24822
|
+
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV × batchSize must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
24803
24823
|
};
|
|
24804
24824
|
}
|
|
24805
24825
|
function handleEstimateApiVsSelfHost(input) {
|
package/dist/index.js
CHANGED
|
@@ -20404,7 +20404,7 @@ function calculate(input) {
|
|
|
20404
20404
|
const computeCeiling = prefillTokensPerSec;
|
|
20405
20405
|
const batchCrossover = quant.bytesPerParam * gpuFlops * 1000 * ETA_COMPUTE / (2 * gpu.memBandwidthGbps * ETA_MEM * quantEff * (tp > 1 ? multiGpuEfficiency : 1));
|
|
20406
20406
|
const memoryBoundAggregate = decodeTokensPerSec * input.batchSize * continuousBatchingMultiplier;
|
|
20407
|
-
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) : 0;
|
|
20407
|
+
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) * tp : 0;
|
|
20408
20408
|
const cachePrefixTokens = input.cachePrefixTokens ?? 0;
|
|
20409
20409
|
const cacheHitRate = input.cacheHitRate ?? 0;
|
|
20410
20410
|
const cacheTTL = input.cacheTTL ?? "none";
|
|
@@ -20868,7 +20868,8 @@ var CompareGpusSchema = object({
|
|
|
20868
20868
|
promptTokens: number2().int().min(1).max(2000000).default(500),
|
|
20869
20869
|
outputTokens: number2().int().min(1).max(200000).default(200),
|
|
20870
20870
|
sortBy: _enum(["lowest_cost", "highest_throughput", "best_value"]).default("best_value"),
|
|
20871
|
-
limit: number2().int().min(1).max(30).default(10)
|
|
20871
|
+
limit: number2().int().min(1).max(30).default(10),
|
|
20872
|
+
gpus: array(string2()).optional().describe("Restrict comparison to specific GPU IDs (e.g. ['h100-sxm','h200-sxm']). Use list_gpus first to find IDs.")
|
|
20872
20873
|
});
|
|
20873
20874
|
var RecommendTopologySchema = object({
|
|
20874
20875
|
model: ModelIdSchema,
|
|
@@ -20973,7 +20974,25 @@ function handleCompareGpus(input) {
|
|
|
20973
20974
|
const quant = QUANT_MAP[input.quantization];
|
|
20974
20975
|
if (!model || !quant)
|
|
20975
20976
|
return { error: "Unknown model or quantization", models: Object.keys(MODEL_MAP).slice(0, 10) };
|
|
20976
|
-
|
|
20977
|
+
let candidatePool = GPUS;
|
|
20978
|
+
if (input.gpus && input.gpus.length > 0) {
|
|
20979
|
+
const requested = new Set(input.gpus);
|
|
20980
|
+
candidatePool = GPUS.filter((g) => requested.has(g.id));
|
|
20981
|
+
if (candidatePool.length === 0) {
|
|
20982
|
+
return {
|
|
20983
|
+
error: `No GPUs matched gpus=[${input.gpus.join(", ")}]. Call list_gpus to find valid IDs.`,
|
|
20984
|
+
validGpuIds: GPUS.slice(0, 10).map((g) => g.id)
|
|
20985
|
+
};
|
|
20986
|
+
}
|
|
20987
|
+
}
|
|
20988
|
+
const skippedNoPrice = [];
|
|
20989
|
+
const candidates = candidatePool.filter((g) => {
|
|
20990
|
+
const fitsMemory = g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam;
|
|
20991
|
+
const hasPrice = g.usdPerHour !== null && g.usdPerHour !== undefined && g.usdPerHour > 0;
|
|
20992
|
+
if (!hasPrice)
|
|
20993
|
+
skippedNoPrice.push(g.id);
|
|
20994
|
+
return fitsMemory && hasPrice;
|
|
20995
|
+
});
|
|
20977
20996
|
const results = candidates.map((g) => {
|
|
20978
20997
|
const r = calculate({
|
|
20979
20998
|
modelId: input.model,
|
|
@@ -21006,10 +21025,11 @@ function handleCompareGpus(input) {
|
|
|
21006
21025
|
return b.aggregateTokensPerSecond / Math.max(b.costPerMillionTokens, 0.001) - a.aggregateTokensPerSecond / Math.max(a.costPerMillionTokens, 0.001);
|
|
21007
21026
|
}).slice(0, input.limit);
|
|
21008
21027
|
return {
|
|
21009
|
-
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens}/M tokens.`,
|
|
21028
|
+
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens.toLocaleString("en-US")}/M tokens.`,
|
|
21010
21029
|
comparisons: sorted,
|
|
21011
21030
|
totalCandidates: results.length,
|
|
21012
21031
|
sortBy: input.sortBy,
|
|
21032
|
+
skippedNoPrice: skippedNoPrice.length > 0 ? skippedNoPrice : undefined,
|
|
21013
21033
|
assumptions: [`Region: us-east-1 (default)`, `On-demand pricing`, `η_mem = 0.65`, `Catalog version: 0.3.0`]
|
|
21014
21034
|
};
|
|
21015
21035
|
}
|
|
@@ -21018,10 +21038,10 @@ function handleRecommendTopology(input) {
|
|
|
21018
21038
|
const quant = QUANT_MAP[input.quantization];
|
|
21019
21039
|
if (!model || !quant)
|
|
21020
21040
|
return { error: "Unknown model or quantization" };
|
|
21041
|
+
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, input.batchSize);
|
|
21021
21042
|
const results = GPUS.map((g) => {
|
|
21022
21043
|
const rec = recommendTopology(model, g, input.contextTokens, input.batchSize, quant.bytesPerParam);
|
|
21023
|
-
const maxConcurrent = computeMaxConcurrency(model, g,
|
|
21024
|
-
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, 1);
|
|
21044
|
+
const maxConcurrent = computeMaxConcurrency(model, g, input.batchSize, input.contextTokens, quant.bytesPerParam);
|
|
21025
21045
|
return {
|
|
21026
21046
|
gpuId: g.id,
|
|
21027
21047
|
gpuName: g.name,
|
|
@@ -21035,12 +21055,12 @@ function handleRecommendTopology(input) {
|
|
|
21035
21055
|
};
|
|
21036
21056
|
}).filter((r) => r.fits).slice(0, 5);
|
|
21037
21057
|
return {
|
|
21038
|
-
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context: ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
21058
|
+
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context (batch=${input.batchSize}): ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
21039
21059
|
model: { name: model.name, paramsB: model.paramsB, activeParamsB: model.activeParamsB, isMoE: model.isMoE },
|
|
21040
|
-
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens) },
|
|
21060
|
+
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens), batchSize: input.batchSize },
|
|
21041
21061
|
recommendations: results,
|
|
21042
|
-
formula: `KV per
|
|
21043
|
-
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
21062
|
+
formula: `KV per batch = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens × ${input.batchSize} batch = ${kvPerRequest.toFixed(2)} GB`,
|
|
21063
|
+
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV × batchSize must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
21044
21064
|
};
|
|
21045
21065
|
}
|
|
21046
21066
|
function handleEstimateApiVsSelfHost(input) {
|