@tokcalc/mcp-server 0.1.3 → 0.2.0-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/README.md +86 -406
  2. package/dist/http.js +23504 -0
  3. package/dist/index.js +21290 -0
  4. package/package.json +36 -92
  5. package/.zscripts/build.sh +0 -175
  6. package/.zscripts/database-runtime-build.sh +0 -33
  7. package/.zscripts/dev.pid +0 -1
  8. package/.zscripts/dev.sh +0 -154
  9. package/.zscripts/mini-services-build.sh +0 -78
  10. package/.zscripts/mini-services-install.sh +0 -65
  11. package/.zscripts/mini-services-start.sh +0 -123
  12. package/.zscripts/python-runtime-build.sh +0 -120
  13. package/.zscripts/start.sh +0 -145
  14. package/CAPACITY_STUDY.md +0 -283
  15. package/CODE_OF_CONDUCT.md +0 -55
  16. package/CONTRIBUTING.md +0 -177
  17. package/Caddyfile +0 -23
  18. package/LICENSE +0 -204
  19. package/bun.lock +0 -1965
  20. package/components.json +0 -21
  21. package/db/custom.db +0 -0
  22. package/download/README.md +0 -1
  23. package/download/tokcalc-dark-calculator.png +0 -0
  24. package/download/tokcalc-dark-default.png +0 -0
  25. package/download/tokcalc-demo.webm +0 -0
  26. package/download/tokcalc-github-link.png +0 -0
  27. package/download/tokcalc-hydration-fixed.png +0 -0
  28. package/download/tokcalc-issue-resolved.png +0 -0
  29. package/download/tokcalc-light-mode.png +0 -0
  30. package/download/tokcalc-light-reference.png +0 -0
  31. package/download/tokcalc-long-context-qwen.png +0 -0
  32. package/download/tokcalc-long-context.png +0 -0
  33. package/download/tokcalc-og-image-preview.png +0 -0
  34. package/download/tokcalc-phase2-3.png +0 -0
  35. package/download/tokcalc-plain-english.png +0 -0
  36. package/download/tokcalc-preview.png +0 -0
  37. package/download/tokcalc-share-bvb.png +0 -0
  38. package/download/tokcalc-share-feature.png +0 -0
  39. package/download/tokcalc-tab-build-vs-buy.png +0 -0
  40. package/download/tokcalc-tab-calculator.png +0 -0
  41. package/download/tokcalc-tab-reference.png +0 -0
  42. package/eslint.config.mjs +0 -50
  43. package/examples/websocket/frontend.tsx +0 -196
  44. package/examples/websocket/server.ts +0 -138
  45. package/mini-services/.gitkeep +0 -0
  46. package/mini-services/mcp-server/README.md +0 -86
  47. package/mini-services/mcp-server/bun.lock +0 -202
  48. package/mini-services/mcp-server/index.ts +0 -504
  49. package/mini-services/mcp-server/package.json +0 -40
  50. package/next.config.ts +0 -12
  51. package/postcss.config.mjs +0 -5
  52. package/prisma/schema.prisma +0 -32
  53. package/public/google6f58ca6be85fa903.html +0 -1
  54. package/public/logo.svg +0 -29
  55. package/public/manifest.json +0 -51
  56. package/public/og-icon-256.png +0 -0
  57. package/public/og.png +0 -0
  58. package/public/robots.txt +0 -25
  59. package/public/sitemap.xml +0 -23
  60. package/public/tokcalc-demo.gif +0 -0
  61. package/scripts/og-template.html +0 -120
  62. package/scripts/render-og.mjs +0 -43
  63. package/server.json +0 -21
  64. package/src/app/api/pricing/aws/route.ts +0 -186
  65. package/src/app/api/pricing/azure/route.ts +0 -168
  66. package/src/app/api/pricing/gcp/route.ts +0 -230
  67. package/src/app/api/pricing/vast-ai/route.ts +0 -164
  68. package/src/app/api/route.ts +0 -5
  69. package/src/app/compare/h100-vs-h200/layout.tsx +0 -30
  70. package/src/app/compare/h100-vs-h200/page.tsx +0 -328
  71. package/src/app/globals.css +0 -122
  72. package/src/app/layout.tsx +0 -276
  73. package/src/app/page.tsx +0 -2670
  74. package/src/components/azure-live-pricing.tsx +0 -185
  75. package/src/components/benchmark-import.tsx +0 -340
  76. package/src/components/confidence-badge.tsx +0 -116
  77. package/src/components/live-pricing-comparison.tsx +0 -241
  78. package/src/components/theme-provider.tsx +0 -11
  79. package/src/components/theme-toggle.tsx +0 -55
  80. package/src/components/ui/accordion.tsx +0 -66
  81. package/src/components/ui/alert-dialog.tsx +0 -157
  82. package/src/components/ui/alert.tsx +0 -66
  83. package/src/components/ui/aspect-ratio.tsx +0 -11
  84. package/src/components/ui/avatar.tsx +0 -53
  85. package/src/components/ui/badge.tsx +0 -46
  86. package/src/components/ui/breadcrumb.tsx +0 -109
  87. package/src/components/ui/button.tsx +0 -59
  88. package/src/components/ui/calendar.tsx +0 -213
  89. package/src/components/ui/card.tsx +0 -92
  90. package/src/components/ui/carousel.tsx +0 -241
  91. package/src/components/ui/chart.tsx +0 -353
  92. package/src/components/ui/checkbox.tsx +0 -32
  93. package/src/components/ui/collapsible.tsx +0 -33
  94. package/src/components/ui/command.tsx +0 -184
  95. package/src/components/ui/context-menu.tsx +0 -252
  96. package/src/components/ui/dialog.tsx +0 -143
  97. package/src/components/ui/drawer.tsx +0 -135
  98. package/src/components/ui/dropdown-menu.tsx +0 -257
  99. package/src/components/ui/form.tsx +0 -167
  100. package/src/components/ui/hover-card.tsx +0 -44
  101. package/src/components/ui/input-otp.tsx +0 -77
  102. package/src/components/ui/input.tsx +0 -21
  103. package/src/components/ui/label.tsx +0 -24
  104. package/src/components/ui/menubar.tsx +0 -276
  105. package/src/components/ui/navigation-menu.tsx +0 -168
  106. package/src/components/ui/pagination.tsx +0 -127
  107. package/src/components/ui/popover.tsx +0 -48
  108. package/src/components/ui/progress.tsx +0 -31
  109. package/src/components/ui/radio-group.tsx +0 -45
  110. package/src/components/ui/resizable.tsx +0 -56
  111. package/src/components/ui/scroll-area.tsx +0 -58
  112. package/src/components/ui/select.tsx +0 -185
  113. package/src/components/ui/separator.tsx +0 -28
  114. package/src/components/ui/sheet.tsx +0 -139
  115. package/src/components/ui/sidebar.tsx +0 -726
  116. package/src/components/ui/skeleton.tsx +0 -13
  117. package/src/components/ui/slider.tsx +0 -63
  118. package/src/components/ui/sonner.tsx +0 -25
  119. package/src/components/ui/switch.tsx +0 -31
  120. package/src/components/ui/table.tsx +0 -116
  121. package/src/components/ui/tabs.tsx +0 -66
  122. package/src/components/ui/textarea.tsx +0 -18
  123. package/src/components/ui/toast.tsx +0 -129
  124. package/src/components/ui/toaster.tsx +0 -35
  125. package/src/components/ui/toggle-group.tsx +0 -73
  126. package/src/components/ui/toggle.tsx +0 -47
  127. package/src/components/ui/tooltip.tsx +0 -61
  128. package/src/components/vast-ai-live-pricing.tsx +0 -176
  129. package/src/hooks/use-mobile.ts +0 -19
  130. package/src/hooks/use-toast.ts +0 -194
  131. package/src/lib/benchmark-parser-sglang.ts +0 -150
  132. package/src/lib/benchmark-parser-tokcalc.ts +0 -247
  133. package/src/lib/benchmark-parser-trtllm.ts +0 -152
  134. package/src/lib/benchmark-parser-vllm.ts +0 -198
  135. package/src/lib/benchmark-schema.ts +0 -263
  136. package/src/lib/db.ts +0 -13
  137. package/src/lib/engine-presets.ts +0 -183
  138. package/src/lib/price-schema.ts +0 -141
  139. package/src/lib/token-calc.ts +0 -808
  140. package/src/lib/track.ts +0 -31
  141. package/src/lib/url-state.ts +0 -256
  142. package/src/lib/utils.ts +0 -6
  143. package/tailwind.config.ts +0 -64
  144. package/tests/database-runtime-build.sh +0 -75
  145. package/tests/python-runtime-build.sh +0 -64
  146. package/tests/python-runtime-container.sh +0 -31
  147. package/tool-results/bash_1789888171144_2c5381860539.txt +0 -161
  148. package/tool-results/bash_1789888175925_49c53ba3c61b.txt +0 -191
  149. package/tool-results/bash_1789888181202_49c53ba3c61b.txt +0 -191
  150. package/tool-results/bash_1789888195219_4a86a5c91411.txt +0 -200
  151. package/tool-results/bash_1789888203128_6cca13c71b47.txt +0 -199
  152. package/tool-results/bash_1789929256963_2a52aff0d0a8.txt +0 -160
  153. package/tool-results/read_1789888151021_69f58eec6a5b.txt +0 -653
  154. package/tool-results/read_1789888153837_1d3a8bfc2a94.txt +0 -653
  155. package/tool-results/read_1789888163087_ccc406d47505.txt +0 -122
  156. package/tool-results/read_1789888167347_67d1d7c9830a.txt +0 -122
  157. package/tool-results/read_1789929252529_d90e8f383a25.txt +0 -285
  158. package/tsconfig.json +0 -42
  159. package/upload/Pasted Content_1789887800864.txt +0 -652
  160. package/upload/Pasted Content_1789887909561.txt +0 -652
  161. package/upload/Pasted Content_1789887918428.txt +0 -652
  162. package/upload/Pasted Content_1789887959420.txt +0 -652
  163. package/upload/Pasted Content_1789888020485.txt +0 -652
  164. package/upload/Pasted Content_1789888058079.txt +0 -652
  165. package/upload/Pasted Content_1789888885033.txt +0 -686
  166. package/upload/Pasted Content_1789928912741.txt +0 -285
  167. package/upload/Pasted Content_1789928938402.txt +0 -285
  168. package/upload/Pasted Content_1789929160389.txt +0 -285
  169. package/upload/Pasted Content_1789929176660.txt +0 -285
  170. package/upload/issue_vision.json +0 -28
  171. package/upload/pasted_image_1789883175209.png +0 -0
  172. package/upload/pasted_image_1789899056690.png +0 -0
  173. package/upload/pasted_image_1789900371483.png +0 -0
  174. package/upload/pasted_image_1789900472823.png +0 -0
  175. package/upload/pasted_image_1789900490374.png +0 -0
  176. package/upload/pasted_image_1789900585552.png +0 -0
  177. package/upload/pasted_image_1789900606519.png +0 -0
  178. package/upload/pasted_image_1789901598705.png +0 -0
  179. package/upload/pasted_image_1789901613545.png +0 -0
  180. package/upload/pasted_image_1789978382674.png +0 -0
  181. package/upload/pasted_image_1789978392749.png +0 -0
  182. package/upload/pasted_image_1789978474879.png +0 -0
  183. package/upload/pasted_image_1789978523652.png +0 -0
  184. package/upload/pasted_image_1789984219089.png +0 -0
  185. package/upload/pasted_image_1789984491896.png +0 -0
  186. package/upload/pasted_image_1789985017950.png +0 -0
  187. package/upload/pasted_image_1789985036765.png +0 -0
  188. package/upload/pasted_image_1789985049848.png +0 -0
  189. package/upload/pasted_image_1790002427833.png +0 -0
  190. package/upload/pasted_image_1790002659944.png +0 -0
  191. package/upload/pasted_image_1790037038476.png +0 -0
  192. package/upload/screenshot_analysis.json +0 -28
  193. package/upload/vision_output.json +0 -28
@@ -1,504 +0,0 @@
1
- /**
2
- * tokcalc MCP Server — local stdio transport.
3
- *
4
- * Exposes 6 read-only planning tools for AI agents (Cursor, Claude Desktop, Cline):
5
- * 1. estimate_capacity — VRAM/KV/throughput/latency/cost for one config
6
- * 2. compare_gpus — ranked GPU comparison for one workload
7
- * 3. recommend_topology — TP/CP topology recommendation
8
- * 4. estimate_api_vs_self_host — break-even analysis
9
- * 5. list_models — discover supported model IDs
10
- * 6. list_gpus — discover supported GPU IDs
11
- */
12
-
13
- import { Server } from "@modelcontextprotocol/sdk/server/index.js";
14
- import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
15
- import {
16
- CallToolRequestSchema,
17
- ListToolsRequestSchema,
18
- } from "@modelcontextprotocol/sdk/types.js";
19
- import { z } from "zod";
20
- import { zodToJsonSchema } from "zod-to-json-schema";
21
-
22
- // Import tokcalc's calculation engine + catalog (shared with the web app)
23
- import {
24
- calculate,
25
- fmtTokens,
26
- fmtBytes,
27
- fmtMs,
28
- fmtMoney,
29
- GPUS,
30
- MODELS,
31
- QUANT_MAP,
32
- GPU_MAP,
33
- MODEL_MAP,
34
- computeMaxConcurrency,
35
- computeKVCacheGb,
36
- recommendTopology,
37
- fmtContext,
38
- type Quantization,
39
- } from "../../src/lib/token-calc.ts";
40
-
41
- // Helper function to format Zod schema for MCP protocol compliance (Strips $schema meta-tag)
42
- function formatInputSchema(schema: z.ZodTypeAny) {
43
- const json = zodToJsonSchema(schema, {
44
- target: "jsonSchema7",
45
- $refStrategy: "none",
46
- }) as Record<string, any>;
47
-
48
- delete json["$schema"];
49
-
50
- return {
51
- type: "object",
52
- properties: json.properties || {},
53
- required: json.required || [],
54
- ...json,
55
- };
56
- }
57
-
58
- // ============================================================
59
- // TOOL SCHEMAS
60
- // ============================================================
61
-
62
- const ModelIdSchema = z.string().describe("Canonical tokcalc model ID (e.g. 'llama3-8b'). Call list_models first if unknown.");
63
- const GpuIdSchema = z.string().describe("Canonical tokcalc GPU ID (e.g. 'h100-sxm'). Call list_gpus first if unknown.");
64
- const QuantSchema = z.enum(["fp32","fp16","bf16","int8","int4","gguf-q2k","gguf-q3km","gguf-q4km","gguf-q5km","gguf-q6k","gguf-q8","gptq4","awq4","exl2-6bpw","fp8","nvfp4"]);
65
-
66
- const EstimateCapacitySchema = z.object({
67
- model: ModelIdSchema,
68
- gpu: GpuIdSchema,
69
- quantization: QuantSchema.default("fp16"),
70
- gpuCount: z.number().int().min(1).max(256).default(1),
71
- batchSize: z.number().int().min(1).max(10000).default(1),
72
- promptTokens: z.number().int().min(1).max(2000000).default(500),
73
- outputTokens: z.number().int().min(1).max(200000).default(200),
74
- engine: z.enum(["generic","vllm","sglang","trtllm","llamacpp"]).default("generic"),
75
- continuousBatching: z.boolean().default(false),
76
- continuousBatchingMultiplier: z.number().min(1).max(10).default(1.5),
77
- reasoningTokens: z.number().int().min(0).max(1000000).default(0),
78
- });
79
-
80
- const CompareGpusSchema = z.object({
81
- model: ModelIdSchema,
82
- quantization: QuantSchema.default("fp16"),
83
- gpuCount: z.number().int().min(1).max(8).default(1),
84
- batchSize: z.number().int().min(1).max(1000).default(8),
85
- promptTokens: z.number().int().min(1).max(2000000).default(500),
86
- outputTokens: z.number().int().min(1).max(200000).default(200),
87
- sortBy: z.enum(["lowest_cost","highest_throughput","best_value"]).default("best_value"),
88
- limit: z.number().int().min(1).max(30).default(10),
89
- });
90
-
91
- const RecommendTopologySchema = z.object({
92
- model: ModelIdSchema,
93
- quantization: QuantSchema.default("fp16"),
94
- contextTokens: z.number().int().min(1024).max(2000000).default(8192),
95
- batchSize: z.number().int().min(1).max(1000).default(1),
96
- });
97
-
98
- const EstimateApiVsSelfHostSchema = z.object({
99
- model: ModelIdSchema,
100
- gpu: GpuIdSchema,
101
- quantization: QuantSchema.default("fp16"),
102
- gpuCount: z.number().int().min(1).max(64).default(1),
103
- inputTokens: z.number().int().min(1).max(2000000).default(500),
104
- outputTokens: z.number().int().min(1).max(200000).default(200),
105
- requestsPerDay: z.number().int().min(1).max(100000000).default(1000),
106
- utilization: z.number().min(0.05).max(1).default(0.5),
107
- apiModel: z.string().default("gpt-4o-mini"),
108
- apiInputPrice: z.number().min(0).default(0.15),
109
- apiOutputPrice: z.number().min(0).default(0.60),
110
- });
111
-
112
- const ListModelsSchema = z.object({
113
- family: z.string().optional().describe("Filter by model family (e.g. 'Llama', 'Qwen')"),
114
- category: z.enum(["text","vlm","embedding","code","reasoning"]).optional(),
115
- isMoE: z.boolean().optional(),
116
- });
117
-
118
- const ListGpusSchema = z.object({
119
- vendor: z.string().optional().describe("Filter by vendor (e.g. 'NVIDIA', 'AMD')"),
120
- category: z.enum(["datacenter","workstation","consumer","mac","tpu","lpu","wse","legacy"]).optional(),
121
- minVramGb: z.number().optional(),
122
- });
123
-
124
- // ============================================================
125
- // TOOL HANDLERS
126
- // ============================================================
127
-
128
- function handleEstimateCapacity(input: z.infer<typeof EstimateCapacitySchema>) {
129
- const result = calculate({
130
- modelId: input.model,
131
- gpuId: input.gpu,
132
- quantization: input.quantization as Quantization,
133
- numGpus: input.gpuCount,
134
- batchSize: input.batchSize,
135
- promptTokens: input.promptTokens,
136
- outputTokens: input.outputTokens,
137
- engineId: input.engine,
138
- effMem: undefined,
139
- useContinuousBatching: input.continuousBatching,
140
- continuousBatchingMultiplier: input.continuousBatchingMultiplier,
141
- reasoningTokens: input.reasoningTokens,
142
- });
143
-
144
- const model = MODEL_MAP[input.model];
145
- const gpu = GPU_MAP[input.gpu];
146
-
147
- return {
148
- summary: `${model?.name || input.model} on ${gpu?.name || input.gpu} (${input.quantization}): ${fmtTokens(result.decodeTokensPerSec)} tok/s decode, ${fmtMs(result.ttftMs)} TTFT, ${fmtBytes(result.totalVramNeededGb)} VRAM needed, ${fmtMoney(result.costPerMillionOutputTokens)}/M tokens`,
149
- feasibility: {
150
- modelFits: result.vramFits,
151
- fitsWithKvCache: result.vramFits,
152
- blockingReasons: result.vramFits ? [] : [`Needs ${fmtBytes(result.totalVramNeededGb)} but only ${gpu?.vramGb * input.gpuCount} GB available`],
153
- warnings: result.longContextWarning ? [result.longContextWarning] : [],
154
- },
155
- performance: {
156
- outputTokensPerSecond: {
157
- low: Math.round(result.decodeTokensPerSec * 0.7),
158
- expected: Math.round(result.decodeTokensPerSec),
159
- high: Math.round(result.decodeTokensPerSec * 1.3),
160
- },
161
- aggregateTokensPerSecond: Math.round(result.aggregateTokensPerSec),
162
- ttftMs: { expected: Math.round(result.ttftMs) },
163
- itlMs: { expected: Math.round(result.itlMs) },
164
- totalLatencyMs: Math.round(result.totalLatencyMs),
165
- },
166
- memory: {
167
- modelWeightsGb: +result.modelSizeGb.toFixed(2),
168
- kvCacheGb: +result.kvCacheTotalGb.toFixed(2),
169
- totalRequiredGb: +result.totalVramNeededGb.toFixed(2),
170
- availableGb: gpu ? gpu.vramGb * input.gpuCount : 0,
171
- utilizationPct: +((result.totalVramNeededGb / ((gpu?.vramGb || 1) * input.gpuCount)) * 100).toFixed(1),
172
- },
173
- cost: {
174
- gpuHourlyUsd: result.costPerHour,
175
- costPerMillionTokens: result.costPerMillionOutputTokens,
176
- costPerRequest: result.costPerRequest,
177
- },
178
- confidence: {
179
- throughput: result.confidence.decodeTokensPerSec,
180
- latency: result.confidence.totalLatencyMs,
181
- memory: result.confidence.totalVramNeededGb,
182
- },
183
- assumptions: [
184
- `η_mem = 0.65 (typical real-world memory utilization)`,
185
- `η_compute = 0.50 (typical compute utilization)`,
186
- `KV cache in FP16 (2 bytes per value)`,
187
- `Engine: ${input.engine} (affects efficiency factors)`,
188
- `Continuous batching: ${input.continuousBatching ? `${input.continuousBatchingMultiplier}× multiplier` : "disabled"}`,
189
- `These are planning estimates, not deployment guarantees`,
190
- ],
191
- catalogVersion: "0.3.0",
192
- };
193
- }
194
-
195
- function handleCompareGpus(input: z.infer<typeof CompareGpusSchema>) {
196
- const model = MODEL_MAP[input.model];
197
- const quant = QUANT_MAP[input.quantization as Quantization];
198
- if (!model || !quant) return { error: "Unknown model or quantization", models: Object.keys(MODEL_MAP).slice(0, 10) };
199
-
200
- const candidates = GPUS.filter(g => g.vramGb * input.gpuCount >= model.activeParamsB * quant.bytesPerParam);
201
-
202
- const results = candidates.map(g => {
203
- const r = calculate({
204
- modelId: input.model,
205
- gpuId: g.id,
206
- quantization: input.quantization as Quantization,
207
- numGpus: input.gpuCount,
208
- batchSize: input.batchSize,
209
- promptTokens: input.promptTokens,
210
- outputTokens: input.outputTokens,
211
- });
212
- return {
213
- gpuId: g.id,
214
- gpuName: g.name,
215
- vendor: g.vendor,
216
- vramGb: g.vramGb * input.gpuCount,
217
- modelFits: r.vramFits,
218
- outputTokensPerSecond: Math.round(r.decodeTokensPerSec),
219
- aggregateTokensPerSecond: Math.round(r.aggregateTokensPerSec),
220
- ttftMs: Math.round(r.ttftMs),
221
- costPerMillionTokens: +r.costPerMillionOutputTokens.toFixed(2),
222
- gpuHourlyUsd: r.costPerHour,
223
- confidence: r.confidence.decodeTokensPerSec,
224
- };
225
- });
226
-
227
- const sorted = results.sort((a, b) => {
228
- if (input.sortBy === "lowest_cost") return a.costPerMillionTokens - b.costPerMillionTokens;
229
- if (input.sortBy === "highest_throughput") return b.aggregateTokensPerSecond - a.aggregateTokensPerSecond;
230
- return (b.aggregateTokensPerSecond / Math.max(b.costPerMillionTokens, 0.001)) - (a.aggregateTokensPerSecond / Math.max(a.costPerMillionTokens, 0.001));
231
- }).slice(0, input.limit);
232
-
233
- return {
234
- summary: `Compared ${results.length} GPUs for ${model.name} (${quant.label}). Cheapest: ${sorted[0]?.gpuName} at $${sorted[0]?.costPerMillionTokens}/M tokens.`,
235
- comparisons: sorted,
236
- totalCandidates: results.length,
237
- sortBy: input.sortBy,
238
- assumptions: [`Region: us-east-1 (default)`, `On-demand pricing`, `η_mem = 0.65`, `Catalog version: 0.3.0`],
239
- };
240
- }
241
-
242
- function handleRecommendTopology(input: z.infer<typeof RecommendTopologySchema>) {
243
- const model = MODEL_MAP[input.model];
244
- const quant = QUANT_MAP[input.quantization as Quantization];
245
- if (!model || !quant) return { error: "Unknown model or quantization" };
246
-
247
- const results = GPUS.map(g => {
248
- const rec = recommendTopology(model, g, input.contextTokens, input.batchSize, quant.bytesPerParam);
249
- const maxConcurrent = computeMaxConcurrency(model, g, 1, input.contextTokens, quant.bytesPerParam);
250
- const kvPerRequest = computeKVCacheGb(model, input.contextTokens, 1);
251
- return {
252
- gpuId: g.id,
253
- gpuName: g.name,
254
- vramGb: g.vramGb,
255
- topology: rec.topology,
256
- neededGpus: rec.neededGpus,
257
- fits: rec.fits,
258
- maxConcurrentUsers: maxConcurrent,
259
- kvPerRequestGb: +kvPerRequest.toFixed(2),
260
- reason: rec.reason,
261
- };
262
- }).filter(r => r.fits).slice(0, 5);
263
-
264
- return {
265
- summary: `For ${model.name} at ${fmtContext(input.contextTokens)} context: ${results.length} feasible topology options across ${GPUS.length} GPUs.`,
266
- model: { name: model.name, paramsB: model.paramsB, activeParamsB: model.activeParamsB, isMoE: model.isMoE },
267
- context: { tokens: input.contextTokens, label: fmtContext(input.contextTokens) },
268
- recommendations: results,
269
- formula: `KV per request = 2 × ${model.layers} layers × ${model.kvHeads} KV heads × ${model.headDim} head_dim × 2 bytes × ${input.contextTokens} tokens = ${computeKVCacheGb(model, input.contextTokens, 1).toFixed(2)} GB`,
270
- assumptions: [`Single GPU unless TP needed`, `KV cache in FP16`, `Model weights + KV must fit in total VRAM`, `Catalog version: 0.3.0`],
271
- };
272
- }
273
-
274
- function handleEstimateApiVsSelfHost(input: z.infer<typeof EstimateApiVsSelfHostSchema>) {
275
- const sh = calculate({
276
- modelId: input.model,
277
- gpuId: input.gpu,
278
- quantization: input.quantization as Quantization,
279
- numGpus: input.gpuCount,
280
- batchSize: 8,
281
- promptTokens: input.inputTokens,
282
- outputTokens: input.outputTokens,
283
- });
284
-
285
- const gpu = GPU_MAP[input.gpu];
286
- const effGpuPrice = (gpu?.usdPerHour ?? 0) * input.gpuCount;
287
- const effTokens = sh.aggregateTokensPerSec * (input.utilization / 100);
288
- const selfHostCostPerM = effTokens > 0 ? (effGpuPrice / 3600 / effTokens) * 1e6 : Infinity;
289
- const selfHostMonthly = effGpuPrice * 730;
290
-
291
- const apiCostPerRequest = (input.inputTokens / 1e6) * input.apiInputPrice + (input.outputTokens / 1e6) * input.apiOutputPrice;
292
- const apiMonthly = apiCostPerRequest * input.requestsPerDay * 30;
293
- const selfHostMonthlyTotal = selfHostMonthly;
294
-
295
- const breakEven = apiCostPerRequest > 0 ? selfHostMonthly / (30 * apiCostPerRequest) : Infinity;
296
-
297
- const cheaper = selfHostCostPerM < input.apiOutputPrice;
298
- const meetsVolume = input.requestsPerDay > breakEven;
299
-
300
- return {
301
- summary: cheaper && meetsVolume
302
- ? `Self-hosting is cheaper at ${input.requestsPerDay.toLocaleString()} req/day. $${selfHostCostPerM.toFixed(2)}/M vs $${input.apiOutputPrice}/M API.`
303
- : !meetsVolume
304
- ? `Not enough volume. Need ${Math.round(breakEven).toLocaleString()} req/day to break even (currently ${input.requestsPerDay.toLocaleString()}).`
305
- : `API is cheaper. Self-host $${selfHostCostPerM.toFixed(2)}/M vs API $${input.apiOutputPrice}/M.`,
306
- selfHost: {
307
- costPerMillionTokens: +selfHostCostPerM.toFixed(2),
308
- monthlyInfraUsd: +selfHostMonthly.toFixed(2),
309
- monthlyTotalUsd: +selfHostMonthlyTotal.toFixed(2),
310
- utilization: `${input.utilization}%`,
311
- throughput: `${fmtTokens(sh.aggregateTokensPerSec)} tok/s`,
312
- },
313
- api: {
314
- costPerMillionTokens: input.apiOutputPrice,
315
- monthlyUsd: +apiMonthly.toFixed(2),
316
- model: input.apiModel,
317
- },
318
- breakEven: {
319
- requestsPerDay: Math.round(breakEven),
320
- reached: meetsVolume,
321
- explanation: `At ${input.utilization}% utilization with ${input.gpuCount}× ${gpu?.name || input.gpu}, self-hosting breaks even at ${Math.round(breakEven).toLocaleString()} requests/day.`,
322
- },
323
- assumptions: [
324
- `Self-host throughput: ${fmtTokens(sh.aggregateTokensPerSec)} tok/s at ${input.utilization}% utilization`,
325
- `GPU price: $${effGpuPrice}/hr`,
326
- `API pricing: $${input.apiInputPrice}/M input, $${input.apiOutputPrice}/M output`,
327
- `730 hours/month`,
328
- `Catalog version: 0.3.0`,
329
- ],
330
- };
331
- }
332
-
333
- function handleListModels(input: z.infer<typeof ListModelsSchema>) {
334
- let filtered = MODELS;
335
- if (input.family) filtered = filtered.filter(m => m.family === input.family);
336
- if (input.category) filtered = filtered.filter(m => m.category === input.category);
337
- if (input.isMoE !== undefined) filtered = filtered.filter(m => m.isMoE === input.isMoE);
338
-
339
- return {
340
- count: filtered.length,
341
- models: filtered.map(m => ({
342
- id: m.id,
343
- name: m.name,
344
- family: m.family,
345
- category: m.category,
346
- paramsB: m.paramsB,
347
- activeParamsB: m.activeParamsB,
348
- isMoE: m.isMoE,
349
- layers: m.layers,
350
- maxContext: m.maxContext,
351
- })),
352
- catalogVersion: "0.3.0",
353
- };
354
- }
355
-
356
- function handleListGpus(input: z.infer<typeof ListGpusSchema>) {
357
- let filtered = GPUS;
358
- if (input.vendor) filtered = filtered.filter(g => g.vendor === input.vendor);
359
- if (input.category) filtered = filtered.filter(g => g.category === input.category);
360
- if (input.minVramGb) filtered = filtered.filter(g => g.vramGb >= input.minVramGb);
361
-
362
- return {
363
- count: filtered.length,
364
- gpus: filtered.map(g => ({
365
- id: g.id,
366
- name: g.name,
367
- vendor: g.vendor,
368
- category: g.category,
369
- memBandwidthGbps: g.memBandwidthGbps,
370
- flopsTflops: g.flopsTflops,
371
- vramGb: g.vramGb,
372
- nvlinkGbps: g.nvlinkGbps,
373
- usdPerHour: g.usdPerHour,
374
- year: g.year,
375
- note: g.note,
376
- })),
377
- catalogVersion: "0.3.0",
378
- };
379
- }
380
-
381
- // ============================================================
382
- // TOOL DEFINITIONS
383
- // ============================================================
384
-
385
- const TOOL_DEFINITIONS = [
386
- {
387
- name: "estimate_capacity",
388
- description: "Estimate whether an LLM-serving configuration fits in memory and can meet throughput and latency targets. Returns VRAM/KV-cache breakdown, throughput and latency ranges, concurrency, cost, confidence, assumptions, and sources.",
389
- inputSchema: formatInputSchema(EstimateCapacitySchema),
390
- },
391
- {
392
- name: "compare_gpus",
393
- description: "Compare supported GPU or cloud SKU options for the same LLM workload. Use before recommending hardware.",
394
- inputSchema: formatInputSchema(CompareGpusSchema),
395
- },
396
- {
397
- name: "recommend_topology",
398
- description: "Recommend feasible GPU/topology designs including tensor parallelism and context parallel (RingAttention) when needed.",
399
- inputSchema: formatInputSchema(RecommendTopologySchema),
400
- },
401
- {
402
- name: "estimate_api_vs_self_host",
403
- description: "Compare monthly token-based API costs with self-hosted GPU infrastructure under stated utilization assumptions.",
404
- inputSchema: formatInputSchema(EstimateApiVsSelfHostSchema),
405
- },
406
- {
407
- name: "list_models",
408
- description: "List tokcalc-supported model IDs and metadata. Use before estimating if the requested model is ambiguous or unknown.",
409
- inputSchema: formatInputSchema(ListModelsSchema),
410
- },
411
- {
412
- name: "list_gpus",
413
- description: "List GPU and cloud SKU IDs, memory, bandwidth, pricing, and categories.",
414
- inputSchema: formatInputSchema(ListGpusSchema),
415
- },
416
- ];
417
-
418
- // ============================================================
419
- // MCP SERVER SETUP
420
- // ============================================================
421
-
422
- const server = new Server(
423
- { name: "tokcalc", version: "0.1.2" },
424
- {
425
- capabilities: {
426
- tools: {},
427
- },
428
- }
429
- );
430
-
431
- // Handle ListTools
432
- server.setRequestHandler(ListToolsRequestSchema, async () => ({
433
- tools: TOOL_DEFINITIONS,
434
- }));
435
-
436
- // Handle CallTool
437
- server.setRequestHandler(CallToolRequestSchema, async (request) => {
438
- const { name, arguments: args } = request.params;
439
-
440
- try {
441
- let result: unknown;
442
-
443
- switch (name) {
444
- case "estimate_capacity": {
445
- const input = EstimateCapacitySchema.parse(args);
446
- result = handleEstimateCapacity(input);
447
- break;
448
- }
449
- case "compare_gpus": {
450
- const input = CompareGpusSchema.parse(args);
451
- result = handleCompareGpus(input);
452
- break;
453
- }
454
- case "recommend_topology": {
455
- const input = RecommendTopologySchema.parse(args);
456
- result = handleRecommendTopology(input);
457
- break;
458
- }
459
- case "estimate_api_vs_self_host": {
460
- const input = EstimateApiVsSelfHostSchema.parse(args);
461
- result = handleEstimateApiVsSelfHost(input);
462
- break;
463
- }
464
- case "list_models": {
465
- const input = ListModelsSchema.parse(args);
466
- result = handleListModels(input);
467
- break;
468
- }
469
- case "list_gpus": {
470
- const input = ListGpusSchema.parse(args);
471
- result = handleListGpus(input);
472
- break;
473
- }
474
- default:
475
- return {
476
- content: [{ type: "text", text: `Unknown tool: ${name}` }],
477
- isError: true,
478
- };
479
- }
480
-
481
- return {
482
- content: [
483
- { type: "text", text: JSON.stringify(result, null, 2) },
484
- ],
485
- structuredContent: result,
486
- };
487
- } catch (error) {
488
- return {
489
- content: [
490
- { type: "text", text: `Error: ${error instanceof Error ? error.message : String(error)}` },
491
- ],
492
- isError: true,
493
- };
494
- }
495
- });
496
-
497
- // ============================================================
498
- // START SERVER (stdio transport)
499
- // ============================================================
500
-
501
- const transport = new StdioServerTransport();
502
- await server.connect(transport);
503
-
504
- console.error("tokcalc MCP server started (stdio transport) — 6 tools available");
@@ -1,40 +0,0 @@
1
- {
2
- "name": "@tokcalc/mcp-server",
3
- "version": "0.1.2",
4
- "description": "tokcalc MCP server — open-source LLM serving capacity planner for AI agents (Cursor, Claude Desktop, Cline). 6 read-only tools: estimate_capacity, compare_gpus, recommend_topology, estimate_api_vs_self_host, list_models, list_gpus.",
5
- "type": "module",
6
- "license": "Apache-2.0",
7
- "bin": {
8
- "tokcalc-mcp": "./dist/index.js",
9
- "tokcalc-mcp-server": "./dist/index.js",
10
- "mcp-server": "./dist/index.js"
11
- },
12
- "files": [
13
- "dist/",
14
- "README.md"
15
- ],
16
- "keywords": [
17
- "mcp",
18
- "model-context-protocol",
19
- "llm",
20
- "inference",
21
- "capacity-planner",
22
- "gpu",
23
- "vllm",
24
- "h100",
25
- "h200",
26
- "tokcalc"
27
- ],
28
- "repository": {
29
- "type": "git",
30
- "url": "https://github.com/stevecrates489-commits/tokcalc"
31
- },
32
- "homepage": "https://tokcalc.vercel.app",
33
- "engines": {
34
- "node": ">=18"
35
- },
36
- "dependencies": {
37
- "@modelcontextprotocol/sdk": "^1.30.1",
38
- "zod-to-json-schema": "^3.25.2"
39
- }
40
- }
package/next.config.ts DELETED
@@ -1,12 +0,0 @@
1
- import type { NextConfig } from "next";
2
-
3
- const nextConfig: NextConfig = {
4
- output: "standalone",
5
- /* config options here */
6
- typescript: {
7
- ignoreBuildErrors: true,
8
- },
9
- reactStrictMode: false,
10
- };
11
-
12
- export default nextConfig;
@@ -1,5 +0,0 @@
1
- const config = {
2
- plugins: ["@tailwindcss/postcss"],
3
- };
4
-
5
- export default config;
@@ -1,32 +0,0 @@
1
- // This is your Prisma schema file,
2
- // learn more about it in the docs: https://pris.ly/d/prisma-schema
3
-
4
- // Looking for ways to speed up your queries, or scale easily with your serverless or edge functions?
5
- // Try Prisma Accelerate: https://pris.ly/cli/accelerate-init
6
-
7
- generator client {
8
- provider = "prisma-client-js"
9
- }
10
-
11
- datasource db {
12
- provider = "sqlite"
13
- url = env("DATABASE_URL")
14
- }
15
-
16
- model User {
17
- id String @id @default(cuid())
18
- email String @unique
19
- name String?
20
- createdAt DateTime @default(now())
21
- updatedAt DateTime @updatedAt
22
- }
23
-
24
- model Post {
25
- id String @id @default(cuid())
26
- title String
27
- content String?
28
- published Boolean @default(false)
29
- authorId String
30
- createdAt DateTime @default(now())
31
- updatedAt DateTime @updatedAt
32
- }
@@ -1 +0,0 @@
1
- google-site-verification: google6f58ca6be85fa903.html
package/public/logo.svg DELETED
@@ -1,29 +0,0 @@
1
- <?xml version="1.0" encoding="utf-8"?>
2
- <svg version="1.1" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" x="0px" y="0px"
3
- viewBox="0 0 30 30" style="enable-background:new 0 0 30 30;" xml:space="preserve">
4
- <defs>
5
- <style type="text/css">
6
- .st194{fill:#2D2D2D;stroke:#FFFFFF;stroke-width:0.6317;stroke-miterlimit:10;}
7
- .st23{fill:#FFFFFF;}
8
-
9
- .z-breathe {
10
- animation: breathe 2.5s ease-in-out infinite;
11
- }
12
-
13
- @keyframes breathe {
14
- 0%, 100% { opacity: 0.7; }
15
- 50% { opacity: 1; }
16
- }
17
- </style>
18
- </defs>
19
-
20
- <g>
21
- <path class="st194" d="M24.51,28.51H5.49c-2.21,0-4-1.79-4-4V5.49c0-2.21,1.79-4,4-4h19.03c2.21,0,4,1.79,4,4v19.03
22
- C28.51,26.72,26.72,28.51,24.51,28.51z"/>
23
- <g class="z-breathe">
24
- <path class="st23" d="M15.47,7.1l-1.3,1.85c-0.2,0.29-0.54,0.47-0.9,0.47h-7.1V7.09C6.16,7.1,15.47,7.1,15.47,7.1z"/>
25
- <polygon class="st23" points="24.3,7.1 13.14,22.91 5.7,22.91 16.86,7.1"/>
26
- <path class="st23" d="M14.53,22.91l1.31-1.86c0.2-0.29,0.54-0.47,0.9-0.47h7.09v2.33H14.53z"/>
27
- </g>
28
- </g>
29
- </svg>
@@ -1,51 +0,0 @@
1
- {
2
- "name": "tokcalc — LLM serving capacity planner",
3
- "short_name": "tokcalc",
4
- "description": "Plan your LLM deployment before you rent the GPUs. Open-source capacity planner: model fit, KV-cache, throughput, latency, multi-GPU scaling, cost — with transparent formulas.",
5
- "start_url": "/",
6
- "display": "standalone",
7
- "background_color": "#0a0e0a",
8
- "theme_color": "#0a0e0a",
9
- "orientation": "any",
10
- "categories": ["developer", "productivity", "utilities"],
11
- "icons": [
12
- {
13
- "src": "/og-icon-256.png",
14
- "sizes": "192x192",
15
- "type": "image/png",
16
- "purpose": "any"
17
- },
18
- {
19
- "src": "/og-icon-256.png",
20
- "sizes": "256x256",
21
- "type": "image/png",
22
- "purpose": "any"
23
- },
24
- {
25
- "src": "/og-icon-256.png",
26
- "sizes": "512x512",
27
- "type": "image/png",
28
- "purpose": "any maskable"
29
- }
30
- ],
31
- "shortcuts": [
32
- {
33
- "name": "Calculator",
34
- "short_name": "Calc",
35
- "description": "Open the LLM capacity calculator",
36
- "url": "/#t=calculator"
37
- },
38
- {
39
- "name": "Build vs Buy",
40
- "short_name": "BvB",
41
- "description": "Compare self-hosting to API pricing",
42
- "url": "/#t=build-vs-buy"
43
- },
44
- {
45
- "name": "Reference catalog",
46
- "short_name": "Ref",
47
- "description": "Browse all models, GPUs, and quantization formats",
48
- "url": "/#t=reference"
49
- }
50
- ]
51
- }
Binary file
package/public/og.png DELETED
Binary file