@tokcalc/mcp-server 0.2.0 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/http.js +30 -10
- package/dist/index.js +30 -10
- package/dist/server.js +30 -10
- package/package.json +5 -5
package/dist/http.js
CHANGED
|
@@ -24163,7 +24163,7 @@ function calculate(input) {
|
|
|
24163
24163
|
const computeCeiling = prefillTokensPerSec;
|
|
24164
24164
|
const batchCrossover = quant.bytesPerParam * gpuFlops * 1000 * ETA_COMPUTE / (2 * gpu.memBandwidthGbps * ETA_MEM * quantEff * (tp > 1 ? multiGpuEfficiency : 1));
|
|
24165
24165
|
const memoryBoundAggregate = decodeTokensPerSec * input.batchSize * continuousBatchingMultiplier;
|
|
24166
|
-
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) : 0;
|
|
24166
|
+
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) * tp : 0;
|
|
24167
24167
|
const cachePrefixTokens = input.cachePrefixTokens ?? 0;
|
|
24168
24168
|
const cacheHitRate = input.cacheHitRate ?? 0;
|
|
24169
24169
|
const cacheTTL = input.cacheTTL ?? "none";
|
|
@@ -24627,7 +24627,8 @@ var CompareGpusSchema = object({
|
|
|
24627
24627
|
promptTokens: number2().int().min(1).max(2000000).default(500),
|
|
24628
24628
|
outputTokens: number2().int().min(1).max(200000).default(200),
|
|
24629
24629
|
sortBy: _enum(["lowest_cost", "highest_throughput", "best_value"]).default("best_value"),
|
|
24630
|
-
limit: number2().int().min(1).max(30).default(10)
|
|
24630
|
+
limit: number2().int().min(1).max(30).default(10),
|
|
24631
|
+
gpus: array(string2()).optional().describe("Restrict comparison to specific GPU IDs (e.g. ['h100-sxm','h200-sxm']). Use list_gpus first to find IDs.")
|
|
24631
24632
|
});
|
|
24632
24633
|
var RecommendTopologySchema = object({
|
|
24633
24634
|
model: ModelIdSchema,
|
|
@@ -24732,7 +24733,25 @@ function handleCompareGpus(input) {
|
|
|
24732
24733
|
const quant = QUANT_MAP[input.quantization];
|
|
24733
24734
|
if (!model || !quant)
|
|
24734
24735
|
return { error: "Unknown model or quantization", models: Object.keys(MODEL_MAP).slice(0, 10) };
|
|
24735
|
-
|
|
24736
|
+
let candidatePool = GPUS;
|
|
24737
|
+
if (input.gpus && input.gpus.length > 0) {
|
|
24738
|
+
const requested = new Set(input.gpus);
|
|
24739
|
+
candidatePool = GPUS.filter((g) => requested.has(g.id));
|
|
24740
|
+
if (candidatePool.length === 0) {
|
|
24741
|
+
return {
|
|
24742
|
+
error: `No GPUs matched gpus=[${input.gpus.join(", ")}]. Call list_gpus to find valid IDs.`,
|
|
24743
|
+
validGpuIds: GPUS.slice(0, 10).map((g) => g.id)
|
|
24744
|
+
};
|
|
24745
|
+
}
|
|
24746
|
+
}
|
|
24747
|
+
const skippedNoPrice = [];
|
|
24748
|
+
const candidates = candidatePool.filter((g) => {
|
|
24749
|
+
const fitsMemory = g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam;
|
|
24750
|
+
const hasPrice = g.usdPerHour !== null && g.usdPerHour !== undefined && g.usdPerHour > 0;
|
|
24751
|
+
if (!hasPrice)
|
|
24752
|
+
skippedNoPrice.push(g.id);
|
|
24753
|
+
return fitsMemory && hasPrice;
|
|
24754
|
+
});
|
|
24736
24755
|
const results = candidates.map((g) => {
|
|
24737
24756
|
const r = calculate({
|
|
24738
24757
|
modelId: input.model,
|
|
@@ -24765,10 +24784,11 @@ function handleCompareGpus(input) {
|
|
|
24765
24784
|
return b.aggregateTokensPerSecond / Math.max(b.costPerMillionTokens, 0.001) - a.aggregateTokensPerSecond / Math.max(a.costPerMillionTokens, 0.001);
|
|
24766
24785
|
}).slice(0, input.limit);
|
|
24767
24786
|
return {
|
|
24768
|
-
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens}/M tokens.`,
|
|
24787
|
+
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens.toLocaleString("en-US")}/M tokens.`,
|
|
24769
24788
|
comparisons: sorted,
|
|
24770
24789
|
totalCandidates: results.length,
|
|
24771
24790
|
sortBy: input.sortBy,
|
|
24791
|
+
skippedNoPrice: skippedNoPrice.length > 0 ? skippedNoPrice : undefined,
|
|
24772
24792
|
assumptions: [`Region: us-east-1 (default)`, `On-demand pricing`, `η_mem = 0.65`, `Catalog version: 0.3.0`]
|
|
24773
24793
|
};
|
|
24774
24794
|
}
|
|
@@ -24777,10 +24797,10 @@ function handleRecommendTopology(input) {
|
|
|
24777
24797
|
const quant = QUANT_MAP[input.quantization];
|
|
24778
24798
|
if (!model || !quant)
|
|
24779
24799
|
return { error: "Unknown model or quantization" };
|
|
24800
|
+
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, input.batchSize);
|
|
24780
24801
|
const results = GPUS.map((g) => {
|
|
24781
24802
|
const rec = recommendTopology(model, g, input.contextTokens, input.batchSize, quant.bytesPerParam);
|
|
24782
|
-
const maxConcurrent = computeMaxConcurrency(model, g,
|
|
24783
|
-
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, 1);
|
|
24803
|
+
const maxConcurrent = computeMaxConcurrency(model, g, input.batchSize, input.contextTokens, quant.bytesPerParam);
|
|
24784
24804
|
return {
|
|
24785
24805
|
gpuId: g.id,
|
|
24786
24806
|
gpuName: g.name,
|
|
@@ -24794,12 +24814,12 @@ function handleRecommendTopology(input) {
|
|
|
24794
24814
|
};
|
|
24795
24815
|
}).filter((r) => r.fits).slice(0, 5);
|
|
24796
24816
|
return {
|
|
24797
|
-
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context: ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
24817
|
+
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context (batch=${input.batchSize}): ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
24798
24818
|
model: { name: model.name, paramsB: model.paramsB, activeParamsB: model.activeParamsB, isMoE: model.isMoE },
|
|
24799
|
-
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens) },
|
|
24819
|
+
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens), batchSize: input.batchSize },
|
|
24800
24820
|
recommendations: results,
|
|
24801
|
-
formula: `KV per
|
|
24802
|
-
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
24821
|
+
formula: `KV per batch = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens × ${input.batchSize} batch = ${kvPerRequest.toFixed(2)} GB`,
|
|
24822
|
+
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV × batchSize must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
24803
24823
|
};
|
|
24804
24824
|
}
|
|
24805
24825
|
function handleEstimateApiVsSelfHost(input) {
|
package/dist/index.js
CHANGED
|
@@ -20404,7 +20404,7 @@ function calculate(input) {
|
|
|
20404
20404
|
const computeCeiling = prefillTokensPerSec;
|
|
20405
20405
|
const batchCrossover = quant.bytesPerParam * gpuFlops * 1000 * ETA_COMPUTE / (2 * gpu.memBandwidthGbps * ETA_MEM * quantEff * (tp > 1 ? multiGpuEfficiency : 1));
|
|
20406
20406
|
const memoryBoundAggregate = decodeTokensPerSec * input.batchSize * continuousBatchingMultiplier;
|
|
20407
|
-
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) : 0;
|
|
20407
|
+
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) * tp : 0;
|
|
20408
20408
|
const cachePrefixTokens = input.cachePrefixTokens ?? 0;
|
|
20409
20409
|
const cacheHitRate = input.cacheHitRate ?? 0;
|
|
20410
20410
|
const cacheTTL = input.cacheTTL ?? "none";
|
|
@@ -20868,7 +20868,8 @@ var CompareGpusSchema = object({
|
|
|
20868
20868
|
promptTokens: number2().int().min(1).max(2000000).default(500),
|
|
20869
20869
|
outputTokens: number2().int().min(1).max(200000).default(200),
|
|
20870
20870
|
sortBy: _enum(["lowest_cost", "highest_throughput", "best_value"]).default("best_value"),
|
|
20871
|
-
limit: number2().int().min(1).max(30).default(10)
|
|
20871
|
+
limit: number2().int().min(1).max(30).default(10),
|
|
20872
|
+
gpus: array(string2()).optional().describe("Restrict comparison to specific GPU IDs (e.g. ['h100-sxm','h200-sxm']). Use list_gpus first to find IDs.")
|
|
20872
20873
|
});
|
|
20873
20874
|
var RecommendTopologySchema = object({
|
|
20874
20875
|
model: ModelIdSchema,
|
|
@@ -20973,7 +20974,25 @@ function handleCompareGpus(input) {
|
|
|
20973
20974
|
const quant = QUANT_MAP[input.quantization];
|
|
20974
20975
|
if (!model || !quant)
|
|
20975
20976
|
return { error: "Unknown model or quantization", models: Object.keys(MODEL_MAP).slice(0, 10) };
|
|
20976
|
-
|
|
20977
|
+
let candidatePool = GPUS;
|
|
20978
|
+
if (input.gpus && input.gpus.length > 0) {
|
|
20979
|
+
const requested = new Set(input.gpus);
|
|
20980
|
+
candidatePool = GPUS.filter((g) => requested.has(g.id));
|
|
20981
|
+
if (candidatePool.length === 0) {
|
|
20982
|
+
return {
|
|
20983
|
+
error: `No GPUs matched gpus=[${input.gpus.join(", ")}]. Call list_gpus to find valid IDs.`,
|
|
20984
|
+
validGpuIds: GPUS.slice(0, 10).map((g) => g.id)
|
|
20985
|
+
};
|
|
20986
|
+
}
|
|
20987
|
+
}
|
|
20988
|
+
const skippedNoPrice = [];
|
|
20989
|
+
const candidates = candidatePool.filter((g) => {
|
|
20990
|
+
const fitsMemory = g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam;
|
|
20991
|
+
const hasPrice = g.usdPerHour !== null && g.usdPerHour !== undefined && g.usdPerHour > 0;
|
|
20992
|
+
if (!hasPrice)
|
|
20993
|
+
skippedNoPrice.push(g.id);
|
|
20994
|
+
return fitsMemory && hasPrice;
|
|
20995
|
+
});
|
|
20977
20996
|
const results = candidates.map((g) => {
|
|
20978
20997
|
const r = calculate({
|
|
20979
20998
|
modelId: input.model,
|
|
@@ -21006,10 +21025,11 @@ function handleCompareGpus(input) {
|
|
|
21006
21025
|
return b.aggregateTokensPerSecond / Math.max(b.costPerMillionTokens, 0.001) - a.aggregateTokensPerSecond / Math.max(a.costPerMillionTokens, 0.001);
|
|
21007
21026
|
}).slice(0, input.limit);
|
|
21008
21027
|
return {
|
|
21009
|
-
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens}/M tokens.`,
|
|
21028
|
+
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens.toLocaleString("en-US")}/M tokens.`,
|
|
21010
21029
|
comparisons: sorted,
|
|
21011
21030
|
totalCandidates: results.length,
|
|
21012
21031
|
sortBy: input.sortBy,
|
|
21032
|
+
skippedNoPrice: skippedNoPrice.length > 0 ? skippedNoPrice : undefined,
|
|
21013
21033
|
assumptions: [`Region: us-east-1 (default)`, `On-demand pricing`, `η_mem = 0.65`, `Catalog version: 0.3.0`]
|
|
21014
21034
|
};
|
|
21015
21035
|
}
|
|
@@ -21018,10 +21038,10 @@ function handleRecommendTopology(input) {
|
|
|
21018
21038
|
const quant = QUANT_MAP[input.quantization];
|
|
21019
21039
|
if (!model || !quant)
|
|
21020
21040
|
return { error: "Unknown model or quantization" };
|
|
21041
|
+
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, input.batchSize);
|
|
21021
21042
|
const results = GPUS.map((g) => {
|
|
21022
21043
|
const rec = recommendTopology(model, g, input.contextTokens, input.batchSize, quant.bytesPerParam);
|
|
21023
|
-
const maxConcurrent = computeMaxConcurrency(model, g,
|
|
21024
|
-
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, 1);
|
|
21044
|
+
const maxConcurrent = computeMaxConcurrency(model, g, input.batchSize, input.contextTokens, quant.bytesPerParam);
|
|
21025
21045
|
return {
|
|
21026
21046
|
gpuId: g.id,
|
|
21027
21047
|
gpuName: g.name,
|
|
@@ -21035,12 +21055,12 @@ function handleRecommendTopology(input) {
|
|
|
21035
21055
|
};
|
|
21036
21056
|
}).filter((r) => r.fits).slice(0, 5);
|
|
21037
21057
|
return {
|
|
21038
|
-
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context: ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
21058
|
+
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context (batch=${input.batchSize}): ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
21039
21059
|
model: { name: model.name, paramsB: model.paramsB, activeParamsB: model.activeParamsB, isMoE: model.isMoE },
|
|
21040
|
-
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens) },
|
|
21060
|
+
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens), batchSize: input.batchSize },
|
|
21041
21061
|
recommendations: results,
|
|
21042
|
-
formula: `KV per
|
|
21043
|
-
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
21062
|
+
formula: `KV per batch = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens × ${input.batchSize} batch = ${kvPerRequest.toFixed(2)} GB`,
|
|
21063
|
+
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV × batchSize must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
21044
21064
|
};
|
|
21045
21065
|
}
|
|
21046
21066
|
function handleEstimateApiVsSelfHost(input) {
|
package/dist/server.js
CHANGED
|
@@ -20297,7 +20297,7 @@ function calculate(input) {
|
|
|
20297
20297
|
const computeCeiling = prefillTokensPerSec;
|
|
20298
20298
|
const batchCrossover = quant.bytesPerParam * gpuFlops * 1000 * ETA_COMPUTE / (2 * gpu.memBandwidthGbps * ETA_MEM * quantEff * (tp > 1 ? multiGpuEfficiency : 1));
|
|
20299
20299
|
const memoryBoundAggregate = decodeTokensPerSec * input.batchSize * continuousBatchingMultiplier;
|
|
20300
|
-
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) : 0;
|
|
20300
|
+
const aggregateTokensPerSec = vramFits ? Math.min(memoryBoundAggregate, computeCeiling) * tp : 0;
|
|
20301
20301
|
const cachePrefixTokens = input.cachePrefixTokens ?? 0;
|
|
20302
20302
|
const cacheHitRate = input.cacheHitRate ?? 0;
|
|
20303
20303
|
const cacheTTL = input.cacheTTL ?? "none";
|
|
@@ -20761,7 +20761,8 @@ var CompareGpusSchema = object2({
|
|
|
20761
20761
|
promptTokens: number2().int().min(1).max(2000000).default(500),
|
|
20762
20762
|
outputTokens: number2().int().min(1).max(200000).default(200),
|
|
20763
20763
|
sortBy: _enum(["lowest_cost", "highest_throughput", "best_value"]).default("best_value"),
|
|
20764
|
-
limit: number2().int().min(1).max(30).default(10)
|
|
20764
|
+
limit: number2().int().min(1).max(30).default(10),
|
|
20765
|
+
gpus: array(string2()).optional().describe("Restrict comparison to specific GPU IDs (e.g. ['h100-sxm','h200-sxm']). Use list_gpus first to find IDs.")
|
|
20765
20766
|
});
|
|
20766
20767
|
var RecommendTopologySchema = object2({
|
|
20767
20768
|
model: ModelIdSchema,
|
|
@@ -20866,7 +20867,25 @@ function handleCompareGpus(input) {
|
|
|
20866
20867
|
const quant = QUANT_MAP[input.quantization];
|
|
20867
20868
|
if (!model || !quant)
|
|
20868
20869
|
return { error: "Unknown model or quantization", models: Object.keys(MODEL_MAP).slice(0, 10) };
|
|
20869
|
-
|
|
20870
|
+
let candidatePool = GPUS;
|
|
20871
|
+
if (input.gpus && input.gpus.length > 0) {
|
|
20872
|
+
const requested = new Set(input.gpus);
|
|
20873
|
+
candidatePool = GPUS.filter((g) => requested.has(g.id));
|
|
20874
|
+
if (candidatePool.length === 0) {
|
|
20875
|
+
return {
|
|
20876
|
+
error: `No GPUs matched gpus=[${input.gpus.join(", ")}]. Call list_gpus to find valid IDs.`,
|
|
20877
|
+
validGpuIds: GPUS.slice(0, 10).map((g) => g.id)
|
|
20878
|
+
};
|
|
20879
|
+
}
|
|
20880
|
+
}
|
|
20881
|
+
const skippedNoPrice = [];
|
|
20882
|
+
const candidates = candidatePool.filter((g) => {
|
|
20883
|
+
const fitsMemory = g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam;
|
|
20884
|
+
const hasPrice = g.usdPerHour !== null && g.usdPerHour !== undefined && g.usdPerHour > 0;
|
|
20885
|
+
if (!hasPrice)
|
|
20886
|
+
skippedNoPrice.push(g.id);
|
|
20887
|
+
return fitsMemory && hasPrice;
|
|
20888
|
+
});
|
|
20870
20889
|
const results = candidates.map((g) => {
|
|
20871
20890
|
const r = calculate({
|
|
20872
20891
|
modelId: input.model,
|
|
@@ -20899,10 +20918,11 @@ function handleCompareGpus(input) {
|
|
|
20899
20918
|
return b.aggregateTokensPerSecond / Math.max(b.costPerMillionTokens, 0.001) - a.aggregateTokensPerSecond / Math.max(a.costPerMillionTokens, 0.001);
|
|
20900
20919
|
}).slice(0, input.limit);
|
|
20901
20920
|
return {
|
|
20902
|
-
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens}/M tokens.`,
|
|
20921
|
+
summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens.toLocaleString("en-US")}/M tokens.`,
|
|
20903
20922
|
comparisons: sorted,
|
|
20904
20923
|
totalCandidates: results.length,
|
|
20905
20924
|
sortBy: input.sortBy,
|
|
20925
|
+
skippedNoPrice: skippedNoPrice.length > 0 ? skippedNoPrice : undefined,
|
|
20906
20926
|
assumptions: [`Region: us-east-1 (default)`, `On-demand pricing`, `η_mem = 0.65`, `Catalog version: 0.3.0`]
|
|
20907
20927
|
};
|
|
20908
20928
|
}
|
|
@@ -20911,10 +20931,10 @@ function handleRecommendTopology(input) {
|
|
|
20911
20931
|
const quant = QUANT_MAP[input.quantization];
|
|
20912
20932
|
if (!model || !quant)
|
|
20913
20933
|
return { error: "Unknown model or quantization" };
|
|
20934
|
+
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, input.batchSize);
|
|
20914
20935
|
const results = GPUS.map((g) => {
|
|
20915
20936
|
const rec = recommendTopology(model, g, input.contextTokens, input.batchSize, quant.bytesPerParam);
|
|
20916
|
-
const maxConcurrent = computeMaxConcurrency(model, g,
|
|
20917
|
-
const kvPerRequest = computeKVCacheGb(model, input.contextTokens, 1);
|
|
20937
|
+
const maxConcurrent = computeMaxConcurrency(model, g, input.batchSize, input.contextTokens, quant.bytesPerParam);
|
|
20918
20938
|
return {
|
|
20919
20939
|
gpuId: g.id,
|
|
20920
20940
|
gpuName: g.name,
|
|
@@ -20928,12 +20948,12 @@ function handleRecommendTopology(input) {
|
|
|
20928
20948
|
};
|
|
20929
20949
|
}).filter((r) => r.fits).slice(0, 5);
|
|
20930
20950
|
return {
|
|
20931
|
-
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context: ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
20951
|
+
summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context (batch=${input.batchSize}): ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
|
|
20932
20952
|
model: { name: model.name, paramsB: model.paramsB, activeParamsB: model.activeParamsB, isMoE: model.isMoE },
|
|
20933
|
-
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens) },
|
|
20953
|
+
context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens), batchSize: input.batchSize },
|
|
20934
20954
|
recommendations: results,
|
|
20935
|
-
formula: `KV per
|
|
20936
|
-
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
20955
|
+
formula: `KV per batch = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens × ${input.batchSize} batch = ${kvPerRequest.toFixed(2)} GB`,
|
|
20956
|
+
assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV × batchSize must fit in total VRAM`, `Catalog version: 0.3.0`]
|
|
20937
20957
|
};
|
|
20938
20958
|
}
|
|
20939
20959
|
function handleEstimateApiVsSelfHost(input) {
|
package/package.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tokcalc/mcp-server",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.1",
|
|
4
4
|
"mcpName": "io.github.stevecrates489-commits/tokcalc",
|
|
5
|
-
"description": "tokcalc MCP server — open-source LLM serving capacity planner for AI agents. v0.2.
|
|
5
|
+
"description": "tokcalc MCP server — open-source LLM serving capacity planner for AI agents. v0.2.1: stdio + stateless HTTP transport with bearer API key auth + KV-backed rate limiting + public hosted endpoint. 7 read-only tools: estimate_capacity, compare_gpus, recommend_topology, estimate_api_vs_self_host, list_models, list_gpus, get_mlperf_benchmarks.",
|
|
6
6
|
"type": "module",
|
|
7
7
|
"main": "./dist/index.js",
|
|
8
8
|
"types": "./dist/index.d.ts",
|
|
@@ -51,9 +51,9 @@
|
|
|
51
51
|
"import": "./dist/server.js"
|
|
52
52
|
},
|
|
53
53
|
"./package.json": "./package.json"
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
54
|
+
},
|
|
55
|
+
"scripts": {
|
|
56
|
+
"build": "bun build ./index.ts ./server.ts ./http.ts --outdir ./dist --target node"
|
|
57
57
|
},
|
|
58
58
|
"dependencies": {
|
|
59
59
|
"@modelcontextprotocol/sdk": "^1.30.1",
|