servcalc 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
servcalc/__init__.py ADDED
@@ -0,0 +1,15 @@
1
+ """servcalc: predict single-GPU LLM serving capacity from config.json."""
2
+
3
+ from . import profiles
4
+ from .calibration import Calibration, Curve
5
+ from .estimate import EstimateParams, ServingEstimate, estimate
6
+ from .gpus import GpuSpec
7
+ from .modes import BatchEstimate, DoesNotFitError
8
+ from .profiles import EngineProfile
9
+ from .timing import Range
10
+ from .workload import Workload
11
+
12
+ __version__ = "0.1.0"
13
+ __all__ = ["BatchEstimate", "Calibration", "Curve", "DoesNotFitError", "EngineProfile",
14
+ "EstimateParams", "GpuSpec", "Range", "ServingEstimate", "Workload", "estimate",
15
+ "profiles"]
@@ -0,0 +1,94 @@
1
+ """Efficiency curves that turn work into time. Priors until measured."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from itertools import pairwise
7
+
8
+
9
+ @dataclass(frozen=True)
10
+ class Curve:
11
+ points: tuple[tuple[float, float, float, float], ...] # (x, low, mid, high)
12
+
13
+ def __post_init__(self):
14
+ if not self.points:
15
+ raise ValueError("Curve needs at least one point")
16
+ xs = [p[0] for p in self.points]
17
+ if xs != sorted(xs) or len(set(xs)) != len(xs):
18
+ raise ValueError("Curve x values must be strictly ascending")
19
+ for x, lo, mid, hi in self.points:
20
+ if not (lo <= mid <= hi):
21
+ raise ValueError(f"Curve point at x={x} must satisfy low <= mid <= high")
22
+ if lo <= 0:
23
+ raise ValueError(f"Curve values must be positive, got {lo} at x={x}")
24
+
25
+ def at(self, x: float) -> tuple[float, float, float]:
26
+ pts = self.points
27
+ if x <= pts[0][0]:
28
+ return pts[0][1:]
29
+ if x >= pts[-1][0]:
30
+ return pts[-1][1:]
31
+ for (x0, *a), (x1, *b) in pairwise(pts):
32
+ if x0 <= x <= x1:
33
+ t = (x - x0) / (x1 - x0)
34
+ return tuple(ya + t * (yb - ya) for ya, yb in zip(a, b))
35
+ raise AssertionError("unreachable")
36
+
37
+
38
+ @dataclass(frozen=True)
39
+ class Calibration:
40
+ bw_util: Curve
41
+ mfu: Curve
42
+ overhead_ms: tuple[float, float, float]
43
+ source: str
44
+
45
+ def __post_init__(self):
46
+ lo, mid, hi = self.overhead_ms
47
+ if not (0 <= lo <= mid <= hi):
48
+ raise ValueError("overhead_ms must satisfy 0 <= low <= mid <= high")
49
+
50
+
51
+ def _curve(*pts: tuple[float, float, float]) -> Curve:
52
+ """Build a Curve from (x, low, high) with mid at the midpoint."""
53
+ return Curve(tuple((x, lo, (lo + hi) / 2, hi) for x, lo, hi in pts))
54
+
55
+
56
+ _BW_X = (1, 8, 64, 512)
57
+ _MFU_X = (1, 64, 512, 4096)
58
+
59
+
60
+ def _prior(bw: tuple, mfu: tuple, overhead: tuple[float, float]) -> Calibration:
61
+ lo, hi = overhead
62
+ return Calibration(
63
+ bw_util=_curve(*((x, a, b) for x, (a, b) in zip(_BW_X, bw))),
64
+ mfu=_curve(*((x, a, b) for x, (a, b) in zip(_MFU_X, mfu))),
65
+ overhead_ms=(lo, (lo + hi) / 2, hi),
66
+ source="prior",
67
+ )
68
+
69
+
70
+ _VLLM_MFU = ((0.02, 0.05), (0.1, 0.2), (0.3, 0.45), (0.35, 0.5))
71
+
72
+ PRIORS: dict[str, Calibration] = {
73
+ "vllm_bf16": _prior(((0.5, 0.7), (0.6, 0.8), (0.65, 0.85), (0.65, 0.85)), _VLLM_MFU, (1, 3)),
74
+ "vllm_int4": _prior(((0.35, 0.55), (0.45, 0.65), (0.5, 0.7), (0.5, 0.7)), _VLLM_MFU, (1, 3)),
75
+ "llamacpp_q4": _prior(((0.3, 0.5), (0.35, 0.6), (0.4, 0.65), (0.4, 0.65)),
76
+ ((0.01, 0.03), (0.08, 0.15), (0.2, 0.35), (0.25, 0.4)), (2, 5)),
77
+ "hf_bf16": _prior(((0.4, 0.6), (0.5, 0.7), (0.5, 0.7), (0.5, 0.7)),
78
+ ((0.02, 0.05), (0.1, 0.2), (0.25, 0.4), (0.3, 0.45)), (3, 8)),
79
+ }
80
+ # Every (engine, quant family) a factory can name resolves here. Families: bf16, int4, q4, q8.
81
+ PRIORS["llamacpp_q8"] = PRIORS["llamacpp_q4"]
82
+ PRIORS["llamacpp_bf16"] = PRIORS["llamacpp_q4"]
83
+ PRIORS["llamacpp_int4"] = PRIORS["llamacpp_q4"]
84
+ PRIORS["vllm_q4"] = PRIORS["vllm_int4"]
85
+ PRIORS["vllm_q8"] = PRIORS["vllm_bf16"]
86
+ PRIORS["hf_int4"] = PRIORS["vllm_int4"]
87
+ PRIORS["hf_q4"] = PRIORS["vllm_int4"]
88
+ PRIORS["hf_q8"] = PRIORS["hf_bf16"]
89
+
90
+
91
+ def get(name: str) -> Calibration:
92
+ if name not in PRIORS:
93
+ raise KeyError(f"unknown calibration {name!r}; built-in: {', '.join(sorted(PRIORS))}")
94
+ return PRIORS[name]
servcalc/cli.py ADDED
@@ -0,0 +1,113 @@
1
+ """Command-line interface: `servcalc MODEL --gpu NAME ...` and `servcalc gpus`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import dataclasses
7
+ import json
8
+ import sys
9
+
10
+ from vramcalc import UnsupportedModelError
11
+ from vramcalc import gpus as _vg
12
+
13
+ from . import profiles
14
+ from .estimate import estimate
15
+ from .gpus import SPECS
16
+ from .modes import DoesNotFitError
17
+ from .workload import Workload
18
+
19
+
20
+ def build_parser() -> argparse.ArgumentParser:
21
+ parser = argparse.ArgumentParser(prog="servcalc", description=__doc__)
22
+ sub = parser.add_subparsers(dest="command")
23
+ est = sub.add_parser("estimate", help="predict serving capacity (default when MODEL is given)")
24
+ est.add_argument("model", help="config.json path, model directory, or Hub id")
25
+ est.add_argument("--gpu", required=True, help="GPU name from `servcalc gpus`")
26
+ est.add_argument("--engine", default="vllm", choices=sorted(profiles.BUILTIN))
27
+ est.add_argument("--prompt", type=int, default=1024, help="prompt tokens per request")
28
+ est.add_argument("--gen", type=int, default=512, help="generated tokens per request")
29
+ est.add_argument("--concurrency", default="1,2,4,8,16,32,64,128,256",
30
+ help="comma-separated batch sizes to evaluate")
31
+ est.add_argument("--quant", help="weight format (awq, gptq, fp8, gguf_q4_k_m, ...)")
32
+ est.add_argument("--kv-bits", type=int, help="KV cache bits (8 or 16)")
33
+ est.add_argument("--max-seqs", type=int, help="engine sequence cap (ollama: num_parallel)")
34
+ est.add_argument("--num-ctx", type=int, help="ollama num_ctx")
35
+ est.add_argument("--token-budget", type=int, help="scheduler tokens per step")
36
+ est.add_argument("--no-chunked", action="store_true", help="vllm: disable chunked prefill")
37
+ est.add_argument("--slo-itl", type=float, metavar="MS",
38
+ help="report max users whose mean ITL is within MS")
39
+ est.add_argument("--json", action="store_true")
40
+ sub.add_parser("gpus", help="list GPUs with bandwidth and bf16 TFLOPs")
41
+ return parser
42
+
43
+
44
+ def _profile(a: argparse.Namespace) -> profiles.EngineProfile:
45
+ """Factory defaults plus CLI overrides; quant left None is inferred from the config later."""
46
+ if a.num_ctx is not None and a.engine != "ollama":
47
+ raise ValueError("--num-ctx applies to --engine ollama only")
48
+ if a.no_chunked and a.engine != "vllm":
49
+ raise ValueError("--no-chunked applies to --engine vllm only")
50
+ p = profiles.BUILTIN[a.engine]()
51
+ if a.quant is not None:
52
+ p = profiles.with_quant(p, a.quant)
53
+ overrides = {}
54
+ if a.kv_bits is not None:
55
+ overrides["kv_bits"] = a.kv_bits
56
+ if a.max_seqs is not None:
57
+ overrides["max_seqs"] = a.max_seqs
58
+ if a.num_ctx is not None:
59
+ overrides["kv_prealloc_ctx"] = a.num_ctx
60
+ if a.token_budget is not None:
61
+ overrides["token_budget"] = a.token_budget
62
+ if a.no_chunked:
63
+ overrides.update(token_budget=None, mixes_prefill=False)
64
+ return dataclasses.replace(p, **overrides) if overrides else p
65
+
66
+
67
+ def cmd_estimate(a: argparse.Namespace) -> int:
68
+ conc = tuple(int(x) for x in a.concurrency.split(","))
69
+ e = estimate(a.model, a.gpu, workload=Workload(a.prompt, a.gen, conc), profile=_profile(a))
70
+ if a.json:
71
+ d = e.to_dict()
72
+ if a.slo_itl is not None:
73
+ d["max_users"] = e.max_users(a.slo_itl)
74
+ print(json.dumps(d, indent=2))
75
+ return 0
76
+ print(e.table())
77
+ if a.slo_itl is not None:
78
+ print(f"\nmax users with mean ITL <= {a.slo_itl:g} ms: {e.max_users(a.slo_itl)}")
79
+ for w in e.warnings:
80
+ print(f"warning: {w}")
81
+ return 0
82
+
83
+
84
+ def cmd_gpus(_: argparse.Namespace) -> int:
85
+ print(f"{'GPU':<20} {'VRAM GiB':>9} {'GB/s':>7} {'bf16 TFLOPs':>12}")
86
+ for name, vram in _vg.GPUS.items():
87
+ bw, tf = SPECS[name]
88
+ print(f"{name:<20} {vram / 2**30:>9.0f} {bw:>7.0f} {tf:>12.0f}")
89
+ return 0
90
+
91
+
92
+ def main(argv: list[str] | None = None) -> int:
93
+ argv = list(sys.argv[1:] if argv is None else argv)
94
+ if argv and argv[0] not in ("gpus", "estimate") and not argv[0].startswith("-"):
95
+ argv.insert(0, "estimate")
96
+ parser = build_parser()
97
+ a = parser.parse_args(argv)
98
+ if a.command is None:
99
+ parser.print_help()
100
+ return 2
101
+ try:
102
+ return {"estimate": cmd_estimate, "gpus": cmd_gpus}[a.command](a)
103
+ except DoesNotFitError as exc:
104
+ print(f"servcalc: {exc}", file=sys.stderr)
105
+ return 1
106
+ except (ValueError, KeyError, FileNotFoundError, ImportError, UnsupportedModelError) as exc:
107
+ msg = exc.args[0] if exc.args else str(exc)
108
+ print(f"servcalc: {msg}", file=sys.stderr)
109
+ return 2
110
+
111
+
112
+ if __name__ == "__main__":
113
+ sys.exit(main())
servcalc/estimate.py ADDED
@@ -0,0 +1,160 @@
1
+ """Compose work, timing, modes and profiles into a ServingEstimate."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import dataclasses
6
+ from dataclasses import dataclass
7
+
8
+ from vramcalc import Arch
9
+ from vramcalc.loaders import load_arch, load_config
10
+
11
+ from . import work
12
+ from .calibration import Calibration
13
+ from .calibration import get as get_calibration
14
+ from .gpus import GpuSpec
15
+ from .gpus import resolve as resolve_gpu
16
+ from .modes import BatchEstimate, capacity, point
17
+ from .profiles import EngineProfile, resolve_profile
18
+ from .timing import Range
19
+ from .workload import Workload
20
+
21
+ GB = 10**9
22
+
23
+ ASSUMPTIONS = (
24
+ "steady state; admission queueing excluded",
25
+ "equal request lengths (prompt_len, gen_len) for every request",
26
+ "prefix caching off; no speculative decoding",
27
+ "FCFS prefill scheduling with leftover token budget passed to the next request",
28
+ "total_tok_s counts decode tokens only (the first token comes from prefill)",
29
+ )
30
+
31
+
32
+ @dataclass(frozen=True)
33
+ class EstimateParams:
34
+ arch: Arch
35
+ gpu: GpuSpec
36
+ workload: Workload
37
+ profile: EngineProfile
38
+ calibration_source: str
39
+
40
+
41
+ def _rng(r: Range | None) -> dict | None:
42
+ return None if r is None else dataclasses.asdict(r)
43
+
44
+
45
+ def _fmt(r: Range | None, digits: int = 0) -> str:
46
+ if r is None:
47
+ return "-"
48
+ return f"{r.mid:.{digits}f} [{r.low:.{digits}f}~{r.high:.{digits}f}]"
49
+
50
+
51
+ @dataclass(frozen=True)
52
+ class ServingEstimate:
53
+ memory_capacity_seqs: int
54
+ configured_capacity_seqs: int
55
+ saturating_batch: int | None
56
+ points: tuple[BatchEstimate, ...]
57
+ weight_bytes: int
58
+ weight_stream_bytes: int
59
+ kv_bytes_per_token: int
60
+ params: EstimateParams
61
+ assumptions: tuple[str, ...]
62
+ warnings: tuple[str, ...]
63
+
64
+ def max_users(self, itl_ms: float) -> int:
65
+ ok = [p.batch for p in self.points if p.feasible and p.itl_ms.mid <= itl_ms]
66
+ return max(ok) if ok else 0
67
+
68
+ def to_dict(self) -> dict:
69
+ points = []
70
+ for p in self.points:
71
+ d = dataclasses.asdict(p)
72
+ for k in ("ttft_ms", "itl_ms", "seq_tok_s", "total_tok_s"):
73
+ d[k] = _rng(getattr(p, k))
74
+ points.append(d)
75
+ return {
76
+ "memory_capacity_seqs": self.memory_capacity_seqs,
77
+ "configured_capacity_seqs": self.configured_capacity_seqs,
78
+ "saturating_batch": self.saturating_batch,
79
+ "weight_bytes": self.weight_bytes,
80
+ "weight_stream_bytes": self.weight_stream_bytes,
81
+ "kv_bytes_per_token": self.kv_bytes_per_token,
82
+ "points": points,
83
+ "params": {
84
+ "arch": dataclasses.asdict(self.params.arch),
85
+ "gpu": dataclasses.asdict(self.params.gpu),
86
+ "workload": dataclasses.asdict(self.params.workload),
87
+ "profile": dataclasses.asdict(self.params.profile),
88
+ "calibration_source": self.params.calibration_source,
89
+ },
90
+ "assumptions": list(self.assumptions),
91
+ "warnings": list(self.warnings),
92
+ }
93
+
94
+ def table(self) -> str:
95
+ p = self.params
96
+ sat = f"; saturating batch {self.saturating_batch}" if self.saturating_batch else ""
97
+ head = [
98
+ f"{p.arch.name or 'model'} on {p.gpu.name} via {p.profile.name} ({p.profile.version})",
99
+ (f"weights {self.weight_bytes / GB:.2f} GB resident, "
100
+ f"{self.weight_stream_bytes / GB:.2f} GB streamed per step; "
101
+ f"KV {self.kv_bytes_per_token / 1e3:.1f} KB/token"),
102
+ (f"capacity: memory {self.memory_capacity_seqs} seqs, "
103
+ f"configured {self.configured_capacity_seqs} seqs{sat}"),
104
+ f"calibration: {p.calibration_source} (ranges are low~high)",
105
+ "",
106
+ (f"{'N':>5} {'R':>5} {'ok':<16} {'TTFT ms':>22} {'ITL ms':>22} "
107
+ f"{'seq tok/s':>20} {'total tok/s':>22} bound"),
108
+ ]
109
+ rows = [
110
+ f"{b.batch:>5} {b.resident:>5} "
111
+ f"{('yes' if b.feasible else 'no:' + str(b.infeasible_reason)):<16} "
112
+ f"{_fmt(b.ttft_ms):>22} {_fmt(b.itl_ms, 1):>22} "
113
+ f"{_fmt(b.seq_tok_s, 1):>20} {_fmt(b.total_tok_s):>22} {b.bound}"
114
+ for b in self.points
115
+ ]
116
+ return "\n".join(head + rows)
117
+
118
+
119
+ def _saturating(points: tuple[BatchEstimate, ...]) -> int | None:
120
+ prev = None
121
+ for p in points:
122
+ if prev == "memory" and p.bound == "compute":
123
+ return p.batch
124
+ if p.bound != "overhead":
125
+ prev = p.bound
126
+ return None
127
+
128
+
129
+ def estimate(model, gpu, *, workload: Workload, profile: EngineProfile | str = "vllm",
130
+ calibration: Calibration | None = None,
131
+ weight_bytes: int | None = None) -> ServingEstimate:
132
+ config = None if isinstance(model, Arch) else load_config(model)
133
+ arch = load_arch(model)
134
+ gpu_spec = resolve_gpu(gpu)
135
+ prof, warnings = resolve_profile(profile, config)
136
+ cal = calibration if calibration is not None else get_calibration(prof.calibration)
137
+ if weight_bytes is not None and (
138
+ not isinstance(weight_bytes, int) or isinstance(weight_bytes, bool)
139
+ or weight_bytes <= 0):
140
+ raise ValueError(f"weight_bytes must be a positive integer, got {weight_bytes!r}")
141
+ wts = work.weight_bytes(arch, prof.weight_bits, weight_bytes)
142
+ if weight_bytes is None and prof.quant is not None:
143
+ warnings.append("quantized weight bytes approximate non-linear tensors as 16-bit; "
144
+ "pass weight_bytes to override")
145
+ kv_tok = work.kv_bytes_per_token(arch, prof.kv_bits)
146
+ cap = capacity(gpu_spec, wts.resident, kv_tok, workload, prof)
147
+ points = tuple(point(n, arch, gpu_spec, workload, prof, cal, wts, kv_tok, cap)
148
+ for n in workload.concurrency)
149
+ return ServingEstimate(
150
+ memory_capacity_seqs=cap.memory_capacity,
151
+ configured_capacity_seqs=cap.configured,
152
+ saturating_batch=_saturating(points),
153
+ points=points,
154
+ weight_bytes=wts.resident,
155
+ weight_stream_bytes=wts.stream,
156
+ kv_bytes_per_token=kv_tok,
157
+ params=EstimateParams(arch, gpu_spec, workload, prof, cal.source),
158
+ assumptions=ASSUMPTIONS,
159
+ warnings=tuple(warnings),
160
+ )
servcalc/gpus.py ADDED
@@ -0,0 +1,57 @@
1
+ """GPU specs: vramcalc's capacity table extended with bandwidth and bf16 compute."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ from dataclasses import dataclass
7
+
8
+ from vramcalc import gpus as _vg
9
+
10
+ # (memory bandwidth GB/s, dense bf16 TFLOPs with fp32 accumulate). Vendor spec sheets, no sparsity.
11
+ SPECS: dict[str, tuple[float, float]] = {
12
+ "RTX 3060": (360, 25), "RTX 3070": (448, 40), "RTX 3080": (760, 60), "RTX 3080 Ti": (912, 68),
13
+ "RTX 3090": (936, 71), "RTX 3090 Ti": (1008, 80),
14
+ "RTX 4060": (272, 30), "RTX 4060 Ti": (288, 44), "RTX 4070": (504, 58), "RTX 4070 Ti": (504, 80),
15
+ "RTX 4070 Ti Super": (672, 88), "RTX 4080": (717, 97), "RTX 4080 Super": (736, 104),
16
+ "RTX 4090": (1008, 165),
17
+ "RTX 5070": (672, 62), "RTX 5070 Ti": (896, 88), "RTX 5080": (960, 112), "RTX 5090": (1792, 210),
18
+ "T4": (320, 65), "V100 16GB": (900, 125), "V100 32GB": (900, 125), # no bf16: fp16 tensor peak
19
+ "A10": (600, 125), "A40": (696, 150), "RTX A6000": (768, 155), "RTX 6000 Ada": (960, 364),
20
+ "A100 40GB": (1555, 312), "A100 80GB": (2039, 312), "A800 80GB": (2039, 312),
21
+ "H100 80GB": (3350, 989), "H800 80GB": (3350, 989), "H200": (4800, 989),
22
+ "L4": (300, 121), "L40S": (864, 362), "B200": (8000, 2250),
23
+ }
24
+
25
+
26
+ _NUMBER_HINT = ("gpu must be a name from the built-in table (see `servcalc gpus`) or a GpuSpec "
27
+ "with bandwidth and TFLOPs; a bare number ({gpu!r}) cannot be timed")
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class GpuSpec:
32
+ name: str
33
+ vram_bytes: int
34
+ bandwidth_gbs: float
35
+ bf16_tflops: float
36
+
37
+ def __post_init__(self):
38
+ for f in ("vram_bytes", "bandwidth_gbs", "bf16_tflops"):
39
+ v = getattr(self, f)
40
+ if (not isinstance(v, (int, float)) or isinstance(v, bool)
41
+ or not math.isfinite(v) or v <= 0):
42
+ raise ValueError(f"{f} must be a positive finite number, got {v!r}")
43
+
44
+
45
+ def resolve(gpu) -> GpuSpec:
46
+ if isinstance(gpu, GpuSpec):
47
+ return gpu
48
+ if not isinstance(gpu, str):
49
+ raise ValueError(_NUMBER_HINT.format(gpu=gpu)) # noqa: TRY004
50
+ try:
51
+ name, vram = _vg.lookup(gpu)
52
+ except KeyError:
53
+ if gpu.strip().replace(".", "", 1).isdigit():
54
+ raise ValueError(_NUMBER_HINT.format(gpu=gpu)) from None
55
+ raise
56
+ bw, tf = SPECS[name]
57
+ return GpuSpec(name, vram, float(bw), float(tf))
servcalc/modes.py ADDED
@@ -0,0 +1,170 @@
1
+ """Batching modes (static, continuous steady state) and capacity. Spec sections 5.4 and 7."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ from dataclasses import dataclass
7
+
8
+ from vramcalc import Arch
9
+
10
+ from . import work
11
+ from .calibration import Calibration
12
+ from .gpus import GpuSpec
13
+ from .profiles import EngineProfile
14
+ from .timing import Range, StepTime, step_time
15
+ from .work import Weights, Work
16
+ from .workload import Workload
17
+
18
+
19
+ class DoesNotFitError(ValueError):
20
+ """Weights alone exceed the engine's memory budget; offload is not modelled."""
21
+
22
+
23
+ @dataclass(frozen=True)
24
+ class Capacity:
25
+ usable: float
26
+ seq_kv_peak: int
27
+ memory_capacity: int
28
+ configured: int
29
+
30
+
31
+ @dataclass(frozen=True)
32
+ class BatchEstimate:
33
+ batch: int
34
+ resident: int
35
+ feasible: bool
36
+ infeasible_reason: str | None
37
+ ttft_ms: Range | None
38
+ itl_ms: Range
39
+ seq_tok_s: Range
40
+ total_tok_s: Range
41
+ bound: str
42
+
43
+
44
+ def _roundup(x: int, block: int) -> int:
45
+ return x if block == 0 else math.ceil(x / block) * block
46
+
47
+
48
+ def capacity(gpu: GpuSpec, w_res: int, kv_tok: int, workload: Workload,
49
+ profile: EngineProfile) -> Capacity:
50
+ budget = gpu.vram_bytes * (profile.memory_budget_ratio - profile.workspace_ratio)
51
+ usable = budget - w_res
52
+ if usable <= 0:
53
+ raise DoesNotFitError(
54
+ f"weights ({w_res / 1e9:.2f} GB) exceed the memory budget of {gpu.name} "
55
+ f"({budget / 1e9:.2f} GB usable); CPU offload is not modelled")
56
+ total_len = workload.prompt_len + workload.gen_len
57
+ if profile.kv_prealloc_ctx is not None:
58
+ if profile.kv_prealloc_ctx < total_len:
59
+ raise ValueError(
60
+ f"kv_prealloc_ctx={profile.kv_prealloc_ctx} is shorter than "
61
+ f"prompt_len + gen_len={total_len}; truncation and context shifting "
62
+ "are not modelled")
63
+ seq_kv_peak = kv_tok * profile.kv_prealloc_ctx
64
+ else:
65
+ seq_kv_peak = kv_tok * _roundup(total_len, profile.kv_block)
66
+ mem = int(usable // seq_kv_peak)
67
+ conf = mem if profile.max_seqs is None else min(mem, profile.max_seqs)
68
+ return Capacity(usable, seq_kv_peak, mem, conf)
69
+
70
+
71
+ def _prefill_ctx_bytes(kv_tok: int, prompt_len: int, profile: EngineProfile) -> int:
72
+ if profile.kv_prealloc_ctx is not None:
73
+ return kv_tok * profile.kv_prealloc_ctx
74
+ return kv_tok * _roundup(prompt_len, profile.kv_block)
75
+
76
+
77
+ def _decode_only(arch, n, workload, kv_tok, wts, profile) -> Work:
78
+ return work.decode_step(arch, n, n * workload.mean_decode_ctx, kv_tok, wts.stream,
79
+ profile.kv_realloc)
80
+
81
+
82
+ def _mean(ranges: list[Range]) -> Range:
83
+ k = len(ranges)
84
+ return Range(sum(r.low for r in ranges) / k, sum(r.mid for r in ranges) / k,
85
+ sum(r.high for r in ranges) / k)
86
+
87
+
88
+ def _sum(ranges: list[Range]) -> Range:
89
+ total = ranges[0]
90
+ for r in ranges[1:]:
91
+ total = total + r
92
+ return total
93
+
94
+
95
+ def _finish(n, resident, reasons: list[str], ttft, itl: StepTime, cap: Capacity,
96
+ kv_needed: float, profile) -> BatchEstimate:
97
+ # reason priority: memory, max_seqs, token_budget
98
+ if profile.max_seqs is not None and resident > profile.max_seqs:
99
+ reasons.insert(0, "max_seqs")
100
+ if kv_needed > cap.usable:
101
+ reasons.insert(0, "memory")
102
+ reason = reasons[0] if reasons else None
103
+ seq = itl.ms.invert(1000.0)
104
+ return BatchEstimate(n, resident, reason is None, reason, ttft, itl.ms, seq, seq.scale(n),
105
+ itl.bound)
106
+
107
+
108
+ def point(n: int, arch: Arch, gpu: GpuSpec, workload: Workload, profile: EngineProfile,
109
+ cal: Calibration, wts: Weights, kv_tok: int, cap: Capacity) -> BatchEstimate:
110
+ if profile.batching == "static":
111
+ return _static(n, arch, gpu, workload, profile, cal, wts, kv_tok, cap)
112
+ return _continuous(n, arch, gpu, workload, profile, cal, wts, kv_tok, cap)
113
+
114
+
115
+ def _static(n, arch, gpu, workload, profile, cal, wts, kv_tok, cap) -> BatchEstimate:
116
+ plen, d = workload.prompt_len, workload.decode_steps
117
+ # n independent causal prompts in one forward pass: lm_head once per sequence
118
+ pre = Work(float(wts.stream) + kv_tok * n * plen,
119
+ 2 * work.p_body(arch) * n * plen + 2 * arch.embed_params() * n
120
+ + 2 * arch.num_layers * arch.q * n * plen * (plen + 1))
121
+ ttft = step_time(pre, gpu, cal, n * plen).ms
122
+ steps = [step_time(work.decode_step(arch, n, n * (plen + 1 + j), kv_tok, wts.stream,
123
+ profile.kv_realloc), gpu, cal, n)
124
+ for j in range(d)]
125
+ itl = StepTime(_mean([s.ms for s in steps]), steps[d // 2].bound)
126
+ return _finish(n, n, [], ttft, itl, cap, n * cap.seq_kv_peak, profile)
127
+
128
+
129
+ def _continuous(n, arch, gpu, workload, profile, cal, wts, kv_tok, cap) -> BatchEstimate:
130
+ plen = workload.prompt_len
131
+ lam = n / workload.decode_steps
132
+ base = _decode_only(arch, n, workload, kv_tok, wts, profile)
133
+ s_d = step_time(base, gpu, cal, n)
134
+ reasons: list[str] = []
135
+
136
+ if profile.mixes_prefill:
137
+ budget = profile.token_budget
138
+ c = plen if budget is None else budget - n
139
+ if budget is not None and (c < 1 or n + lam * plen > budget):
140
+ reasons.append("token_budget")
141
+ if c < 1:
142
+ # prefill cannot be represented; report decode-only numbers
143
+ return _finish(n, n, reasons, None, s_d, cap, n * cap.seq_kv_peak, profile)
144
+ chunks = work.prefill_chunks(plen, c)
145
+ k = len(chunks)
146
+ f = min(1.0, lam * plen / c)
147
+ steps = [step_time(base + work.prefill_chunk(arch, clen, c0, last, kv_tok), gpu, cal,
148
+ n + clen)
149
+ for clen, c0, last in chunks]
150
+ ttft = _sum([s.ms for s in steps])
151
+ mixed_mean = ttft.scale(1 / k) # representative mixed step: mean of the real chunk steps
152
+ itl = StepTime(Range.weighted(mixed_mean, f, s_d.ms, 1 - f),
153
+ steps[0].bound if f >= 0.5 else s_d.bound)
154
+ m = math.ceil(lam * k)
155
+ else:
156
+ chunks = work.prefill_chunks(plen, profile.token_budget)
157
+ lm_head = 2 * arch.embed_params() # read only on the chunk that emits the token
158
+ parts = []
159
+ for clen, c0, last in chunks:
160
+ wread = max(0, wts.stream - (0 if last else lm_head))
161
+ parts.append(step_time(Work(float(wread), 0.0)
162
+ + work.prefill_chunk(arch, clen, c0, last, kv_tok),
163
+ gpu, cal, clen).ms)
164
+ t_p = _sum(parts)
165
+ itl = StepTime(s_d.ms + t_p.scale(lam), s_d.bound)
166
+ ttft = t_p + s_d.ms.scale(0.5)
167
+ m = math.ceil(lam * t_p.mid / itl.ms.mid)
168
+
169
+ kv_needed = n * cap.seq_kv_peak + m * _prefill_ctx_bytes(kv_tok, plen, profile)
170
+ return _finish(n, n + m, reasons, ttft, itl, cap, kv_needed, profile)
servcalc/profiles.py ADDED
@@ -0,0 +1,163 @@
1
+ """Engine profiles: everything that differs between serving engines, as values."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import dataclasses
6
+ from dataclasses import dataclass
7
+
8
+ from . import quant as _quant
9
+
10
+ BATCHING = ("continuous", "static")
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class EngineProfile:
15
+ name: str
16
+ version: str
17
+ batching: str
18
+ max_seqs: int | None
19
+ quant: str | None
20
+ kv_bits: int
21
+ kv_block: int
22
+ kv_prealloc_ctx: int | None
23
+ memory_budget_ratio: float
24
+ workspace_ratio: float
25
+ token_budget: int | None
26
+ mixes_prefill: bool
27
+ kv_realloc: bool
28
+ calibration: str
29
+
30
+ def __post_init__(self):
31
+ if self.batching not in BATCHING:
32
+ raise ValueError(f"batching must be one of {BATCHING}, got {self.batching!r}")
33
+ _quant.bits(self.quant) # validates the name
34
+ if self.kv_bits not in (8, 16):
35
+ raise ValueError(f"kv_bits must be 8 or 16, got {self.kv_bits!r}")
36
+ for f in ("max_seqs", "kv_prealloc_ctx", "token_budget"):
37
+ v = getattr(self, f)
38
+ if v is not None and (not isinstance(v, int) or isinstance(v, bool) or v <= 0):
39
+ raise ValueError(f"{f} must be a positive integer or None, got {v!r}")
40
+ if self.token_budget is not None and self.token_budget < 2:
41
+ raise ValueError(
42
+ "token_budget must be >= 2 (one decode token plus one prefill token)")
43
+ if not isinstance(self.kv_block, int) or self.kv_block < 0:
44
+ raise ValueError(f"kv_block must be a non-negative integer, got {self.kv_block!r}")
45
+ for f in ("memory_budget_ratio", "workspace_ratio"):
46
+ v = getattr(self, f)
47
+ if not (0 < v <= 1):
48
+ raise ValueError(f"{f} must be in (0, 1], got {v!r}")
49
+ if self.memory_budget_ratio - self.workspace_ratio <= 0:
50
+ raise ValueError("workspace_ratio must be smaller than memory_budget_ratio")
51
+ if self.batching == "static" and self.mixes_prefill:
52
+ raise ValueError("static batching cannot mix prefill and decode")
53
+
54
+ @property
55
+ def weight_bits(self) -> float:
56
+ return _quant.bits(self.quant)
57
+
58
+
59
+ def _cal(engine: str, quant_name: str | None) -> str:
60
+ return f"{engine}_{_quant.curve_suffix(quant_name)}"
61
+
62
+
63
+ def vllm(chunked: bool = True, **overrides) -> EngineProfile:
64
+ """vLLM 0.6.3. Chunked prefill with a 512-token budget that includes decode tokens."""
65
+ quant_name = overrides.pop("quant", None)
66
+ fields = {
67
+ "name": "vllm", "version": "0.6.3", "batching": "continuous", "max_seqs": 256,
68
+ "quant": quant_name, "kv_bits": 16, "kv_block": 16, "kv_prealloc_ctx": None,
69
+ "memory_budget_ratio": 0.90, "workspace_ratio": 0.05,
70
+ "token_budget": 512 if chunked else None, "mixes_prefill": chunked, "kv_realloc": False,
71
+ "calibration": _cal("vllm", quant_name),
72
+ }
73
+ return EngineProfile(**{**fields, **overrides})
74
+
75
+
76
+ def ollama(num_parallel: int = 4, num_ctx: int = 2048, quant: str | None = "gguf_q4_k_m",
77
+ **overrides) -> EngineProfile:
78
+ """ollama 0.5.x (llama.cpp server). Prefill is modelled as a separate pass; confirm in 0.2."""
79
+ fields = {
80
+ "name": "ollama", "version": "0.5.x (llama.cpp server)", "batching": "continuous",
81
+ "max_seqs": num_parallel, "quant": quant, "kv_bits": 16, "kv_block": 0,
82
+ "kv_prealloc_ctx": num_ctx, "memory_budget_ratio": 1.0, "workspace_ratio": 0.15,
83
+ "token_budget": 512, "mixes_prefill": False, "kv_realloc": False,
84
+ "calibration": _cal("llamacpp", quant),
85
+ }
86
+ return EngineProfile(**{**fields, **overrides})
87
+
88
+
89
+ def hf(**overrides) -> EngineProfile:
90
+ """transformers 4.46.x generate() with the model loaded explicitly in bf16; static batching."""
91
+ quant_name = overrides.pop("quant", None)
92
+ fields = {
93
+ "name": "hf", "version": "transformers 4.46.x, explicit bf16 load", "batching": "static",
94
+ "max_seqs": None, "quant": quant_name, "kv_bits": 16, "kv_block": 0,
95
+ "kv_prealloc_ctx": None, "memory_budget_ratio": 1.0, "workspace_ratio": 0.20,
96
+ "token_budget": None, "mixes_prefill": False, "kv_realloc": True,
97
+ "calibration": _cal("hf", quant_name),
98
+ }
99
+ return EngineProfile(**{**fields, **overrides})
100
+
101
+
102
+ BUILTIN = {"vllm": vllm, "ollama": ollama, "hf": hf}
103
+
104
+
105
+ def with_quant(p: EngineProfile, quant_name: str | None) -> EngineProfile:
106
+ """Copy of p with another weight format and the matching calibration family."""
107
+ engine = p.calibration.rsplit("_", 1)[0]
108
+ return dataclasses.replace(p, quant=quant_name, calibration=_cal(engine, quant_name))
109
+
110
+
111
+ # HF quant_method -> {bits: quant name}
112
+ _HF_QUANT_METHODS = {
113
+ "awq": {4: "awq"},
114
+ "gptq": {4: "gptq", 8: "gptq_8"},
115
+ "fp8": {8: "fp8"},
116
+ }
117
+ _GGUF_ENGINES = ("ollama",)
118
+
119
+
120
+ def _config_quant(config: dict | None) -> str | None:
121
+ """Quant name implied by config.json's quantization_config, or None when absent."""
122
+ qc = (config or {}).get("quantization_config")
123
+ if qc is None:
124
+ return None
125
+ method = qc.get("quant_method")
126
+ if method is None:
127
+ raise ValueError("quantization_config has no quant_method; pass quant explicitly")
128
+ if method not in _HF_QUANT_METHODS:
129
+ raise ValueError(f"unsupported quant_method {method!r}; "
130
+ f"supported: {', '.join(_HF_QUANT_METHODS)}")
131
+ by_bits = _HF_QUANT_METHODS[method]
132
+ bits = qc.get("bits", next(iter(by_bits)))
133
+ if bits not in by_bits:
134
+ raise ValueError(f"{method} with bits={bits!r} is not supported; "
135
+ f"supported bits: {', '.join(map(str, by_bits))}")
136
+ return by_bits[bits]
137
+
138
+
139
+ def resolve_profile(profile, config: dict | None) -> tuple[EngineProfile, list[str]]:
140
+ """Name or EngineProfile into a concrete profile.
141
+
142
+ Precedence for quant: explicit factory argument > config quantization_config > default.
143
+ GGUF engines keep their own format and only warn when the config names an HF quant.
144
+ """
145
+ warnings: list[str] = []
146
+ if isinstance(profile, EngineProfile):
147
+ p = profile
148
+ elif profile in BUILTIN:
149
+ p = BUILTIN[profile]()
150
+ else:
151
+ raise KeyError(f"unknown profile {profile!r}; built-in: {', '.join(BUILTIN)}")
152
+ inferred = _config_quant(config)
153
+ if inferred is not None:
154
+ if p.name in _GGUF_ENGINES:
155
+ warnings.append(
156
+ f"config.json declares {inferred!r} weights but the {p.name} profile models "
157
+ f"quant={p.quant!r}; pass weight_bytes for the GGUF file you serve")
158
+ elif p.quant is None:
159
+ p = with_quant(p, inferred)
160
+ if p.quant is not None:
161
+ warnings.append(f"quant={p.quant!r} sets storage bytes only; "
162
+ f"kernel speed comes from the {p.calibration} curve")
163
+ return p, warnings
servcalc/quant.py ADDED
@@ -0,0 +1,32 @@
1
+ """Weight quantization formats: storage bits per linear weight and the calibration family."""
2
+
3
+ from __future__ import annotations
4
+
5
+ # name -> (bits per linear weight, calibration curve suffix)
6
+ QUANTS: dict[str | None, tuple[float, str]] = {
7
+ None: (16, "bf16"),
8
+ "fp8": (8, "bf16"),
9
+ "awq": (4, "int4"),
10
+ "gptq": (4, "int4"),
11
+ "gptq_8": (8, "q8"),
12
+ "gguf_q8_0": (8.5, "q8"),
13
+ "gguf_q6_k": (6.6, "q4"),
14
+ "gguf_q5_k_m": (5.7, "q4"),
15
+ "gguf_q4_k_m": (4.9, "q4"),
16
+ "gguf_q4_0": (4.5, "q4"),
17
+ }
18
+
19
+
20
+ def _entry(name: str | None) -> tuple[float, str]:
21
+ if name not in QUANTS:
22
+ known = ", ".join(str(k) for k in QUANTS if k)
23
+ raise ValueError(f"unknown quant {name!r}; supported: none, {known}")
24
+ return QUANTS[name]
25
+
26
+
27
+ def bits(name: str | None) -> float:
28
+ return _entry(name)[0]
29
+
30
+
31
+ def curve_suffix(name: str | None) -> str:
32
+ return _entry(name)[1]
servcalc/timing.py ADDED
@@ -0,0 +1,61 @@
1
+ """Turn Work into time with efficiency curves. Spec section 6."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+
7
+ from .calibration import Calibration
8
+ from .gpus import GpuSpec
9
+ from .work import Work
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class Range:
14
+ low: float
15
+ mid: float
16
+ high: float
17
+
18
+ def __post_init__(self):
19
+ if not (self.low <= self.mid <= self.high):
20
+ raise ValueError(f"Range must satisfy low <= mid <= high, got {self}")
21
+
22
+ def __add__(self, o: Range) -> Range:
23
+ return Range(self.low + o.low, self.mid + o.mid, self.high + o.high)
24
+
25
+ def scale(self, k: float) -> Range:
26
+ return Range(self.low * k, self.mid * k, self.high * k)
27
+
28
+ def invert(self, numerator: float) -> Range:
29
+ """numerator / self, swapping ends so that low <= high still holds."""
30
+ return Range(numerator / self.high, numerator / self.mid, numerator / self.low)
31
+
32
+ @staticmethod
33
+ def weighted(a: Range, wa: float, b: Range, wb: float) -> Range:
34
+ return a.scale(wa) + b.scale(wb)
35
+
36
+
37
+ @dataclass(frozen=True)
38
+ class StepTime:
39
+ ms: Range
40
+ bound: str # "memory" | "compute" | "overhead", judged at mid
41
+
42
+
43
+ def step_time(work: Work, gpu: GpuSpec, cal: Calibration, tokens: int) -> StepTime:
44
+ bw = cal.bw_util.at(tokens)
45
+ mfu = cal.mfu.at(tokens)
46
+ oh = cal.overhead_ms
47
+
48
+ def one(util_bw: float, util_mfu: float, overhead: float) -> tuple[float, float, float]:
49
+ t_mem = work.bytes / (gpu.bandwidth_gbs * 1e9 * util_bw) * 1000
50
+ t_comp = work.flops / (gpu.bf16_tflops * 1e12 * util_mfu) * 1000
51
+ return t_mem, t_comp, max(t_mem, t_comp) + overhead
52
+
53
+ # low time uses high efficiency and low overhead; high time the reverse
54
+ _, _, lo = one(bw[2], mfu[2], oh[0])
55
+ t_mem, t_comp, mid = one(bw[1], mfu[1], oh[1])
56
+ _, _, hi = one(bw[0], mfu[0], oh[2])
57
+ if oh[1] > max(t_mem, t_comp):
58
+ bound = "overhead"
59
+ else:
60
+ bound = "memory" if t_mem > t_comp else "compute"
61
+ return StepTime(Range(lo, mid, hi), bound)
servcalc/work.py ADDED
@@ -0,0 +1,84 @@
1
+ """Work formulas: bytes moved and FLOPs per step. Spec section 5. Engine-agnostic."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ from dataclasses import dataclass
7
+
8
+ from vramcalc import Arch
9
+ from vramcalc.memory import kv_cache
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class Work:
14
+ bytes: float
15
+ flops: float
16
+
17
+ def __add__(self, other: Work) -> Work:
18
+ return Work(self.bytes + other.bytes, self.flops + other.flops)
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class Weights:
23
+ resident: int # bytes held in VRAM
24
+ stream: int # bytes read per step (untied input embedding is a lookup, not a scan)
25
+
26
+
27
+ def p_lin(arch: Arch) -> int:
28
+ return arch.num_layers * (arch.attn_params() + arch.mlp_params())
29
+
30
+
31
+ def p_mm(arch: Arch) -> int:
32
+ embed = arch.embed_params()
33
+ return arch.param_count() - (0 if arch.tie_word_embeddings else embed)
34
+
35
+
36
+ def p_body(arch: Arch) -> int:
37
+ return p_mm(arch) - arch.embed_params()
38
+
39
+
40
+ def weight_bytes(arch: Arch, bits: float, override: int | None = None) -> Weights:
41
+ lin = p_lin(arch)
42
+ if override is not None:
43
+ resident = override
44
+ else:
45
+ resident = math.ceil(lin * bits / 8) + (arch.param_count() - lin) * 2
46
+ embed_bytes = 0 if arch.tie_word_embeddings else arch.embed_params() * 2
47
+ return Weights(resident, max(0, resident - embed_bytes))
48
+
49
+
50
+ def kv_bytes_per_token(arch: Arch, kv_bits: int) -> int:
51
+ return kv_cache(arch, 1, 1, 1) * kv_bits // 8
52
+
53
+
54
+ def decode_step(arch: Arch, n: int, sum_ctx: float, kv_tok: int, w_stream: int,
55
+ kv_realloc: bool) -> Work:
56
+ kv_read = kv_tok * sum_ctx
57
+ bytes_ = w_stream + kv_read + kv_tok * n + (2 * kv_read if kv_realloc else 0)
58
+ flops = 2 * p_mm(arch) * n + 4 * arch.num_layers * arch.q * sum_ctx
59
+ return Work(float(bytes_), float(flops))
60
+
61
+
62
+ def prefill_chunk(arch: Arch, c_len: int, prior_ctx: int, last: bool, kv_tok: int) -> Work:
63
+ """One prefill chunk without the weight read (callers add it once per step)."""
64
+ layers, q = arch.num_layers, arch.q
65
+ attn = 4 * layers * q * (c_len * prior_ctx + c_len * (c_len + 1) / 2)
66
+ flops = 2 * p_body(arch) * c_len + attn + (2 * arch.embed_params() if last else 0)
67
+ return Work(float(kv_tok * (prior_ctx + c_len)), float(flops))
68
+
69
+
70
+ def prefill_chunks(prompt_len: int, chunk: int | None) -> list[tuple[int, int, bool]]:
71
+ """(chunk_len, prior_ctx, is_last) for each chunk of a prompt."""
72
+ c = prompt_len if chunk is None else chunk
73
+ k = math.ceil(prompt_len / c)
74
+ out = []
75
+ for j in range(k):
76
+ start = j * c
77
+ out.append((min(c, prompt_len - start), start, j == k - 1))
78
+ return out
79
+
80
+
81
+ def prefill_total_flops(arch: Arch, prompt_len: int) -> float:
82
+ i = prompt_len
83
+ return float(2 * p_body(arch) * i + 2 * arch.embed_params()
84
+ + 2 * arch.num_layers * arch.q * i * (i + 1))
servcalc/workload.py ADDED
@@ -0,0 +1,39 @@
1
+ """Request shape: prompt and generation lengths plus the concurrency sweep."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+
7
+
8
+ def _positive_int(name: str, v) -> None:
9
+ if not isinstance(v, int) or isinstance(v, bool) or v <= 0:
10
+ raise ValueError(f"{name} must be a positive integer, got {v!r}")
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class Workload:
15
+ prompt_len: int
16
+ gen_len: int
17
+ concurrency: tuple[int, ...] = (1, 2, 4, 8, 16, 32, 64, 128, 256)
18
+
19
+ def __post_init__(self):
20
+ _positive_int("prompt_len", self.prompt_len)
21
+ if not isinstance(self.gen_len, int) or isinstance(self.gen_len, bool) or self.gen_len < 2:
22
+ raise ValueError(
23
+ "gen_len must be an integer >= 2 (prefill emits the first token), "
24
+ f"got {self.gen_len!r}")
25
+ if not self.concurrency:
26
+ raise ValueError("concurrency must contain at least one value")
27
+ for n in self.concurrency:
28
+ _positive_int("concurrency", n)
29
+ if list(self.concurrency) != sorted(set(self.concurrency)):
30
+ raise ValueError("concurrency must be strictly ascending")
31
+
32
+ @property
33
+ def decode_steps(self) -> int:
34
+ return self.gen_len - 1
35
+
36
+ @property
37
+ def mean_decode_ctx(self) -> float:
38
+ """Average context during decode: step j has I + 1 + j for 0 <= j < D."""
39
+ return self.prompt_len + 1 + (self.decode_steps - 1) / 2
@@ -0,0 +1,185 @@
1
+ Metadata-Version: 2.5
2
+ Name: servcalc
3
+ Version: 0.1.0
4
+ Summary: Predict single-GPU LLM serving capacity (concurrency, tok/s, TTFT, ITL) from config.json with a roofline model
5
+ License-Expression: MIT
6
+ License-File: LICENSE
7
+ Classifier: License :: OSI Approved :: MIT License
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
10
+ Requires-Python: >=3.10
11
+ Requires-Dist: vramcalc>=0.1.2
12
+ Provides-Extra: dev
13
+ Requires-Dist: pytest; extra == 'dev'
14
+ Requires-Dist: ruff; extra == 'dev'
15
+ Provides-Extra: hf
16
+ Requires-Dist: vramcalc[hf]; extra == 'hf'
17
+ Description-Content-Type: text/markdown
18
+
19
+ # servcalc
20
+
21
+ [![PyPI](https://img.shields.io/pypi/v/servcalc)](https://pypi.org/project/servcalc/) [![CI](https://github.com/sacom123/servcalc/actions/workflows/ci.yml/badge.svg)](https://github.com/sacom123/servcalc/actions/workflows/ci.yml) [![Python](https://img.shields.io/pypi/pyversions/servcalc)](https://pypi.org/project/servcalc/) [![License: MIT](https://img.shields.io/badge/license-MIT-green.svg)](LICENSE)
22
+
23
+ [English](#english) | [한국어](#한국어)
24
+
25
+ ---
26
+
27
+ ## English
28
+
29
+ Predict how many concurrent users one GPU can serve for an LLM, and at what speed, before you rent the GPU. servcalc reads the Hugging Face `config.json`, a GPU name and a serving engine (vLLM, ollama, or plain `transformers`), and returns per batch size: resident sequences, time to first token (TTFT), inter-token latency (ITL), tokens per second, and whether that batch size is feasible on the chosen engine. It is the next question after [vramcalc](https://github.com/sacom123/vramcalc): vramcalc answers "does it fit", servcalc answers "how fast, for how many".
30
+
31
+ ```python
32
+ from servcalc import Workload, estimate
33
+
34
+ e = estimate("meta-llama/Llama-3.1-8B", "RTX 3090", workload=Workload(prompt_len=1024, gen_len=512))
35
+ print(e.table())
36
+ print(e.max_users(itl_ms=50)) # largest feasible batch whose mean ITL is within 50 ms
37
+ for p in e.points:
38
+ print(p.batch, p.feasible, p.itl_ms.mid, p.total_tok_s.mid)
39
+ ```
40
+
41
+ ```
42
+ $ servcalc meta-llama/Llama-3.1-8B --gpu "RTX 3090" --concurrency 1,4,16,64 --slo-itl 50
43
+ Llama-3.1-8B on RTX 3090 via vllm (0.6.3)
44
+ weights 16.06 GB resident, 15.01 GB streamed per step; KV 131.1 KB/token
45
+ capacity: memory 29 seqs, configured 29 seqs; saturating batch 16
46
+ calibration: prior (ranges are low~high)
47
+
48
+ N R ok TTFT ms ITL ms seq tok/s total tok/s bound
49
+ 1 2 yes 579 [482~724] 29.7 [24.7~36.2] 33.7 [27.6~40.5] 34 [28~40] memory
50
+ 4 5 yes 600 [493~768] 30.8 [25.8~43.5] 32.5 [23.0~38.8] 130 [92~155] memory
51
+ 16 17 yes 638 [522~821] 68.6 [50.8~105.2] 14.6 [9.5~19.7] 233 [152~315] compute
52
+ 64 65 no:memory 747 [612~956] 139.9 [109.6~194.3] 7.1 [5.1~9.1] 457 [329~584] compute
53
+
54
+ max users with mean ITL <= 50 ms: 4
55
+ ```
56
+
57
+ `N` is the number of sequences in decode, `R` the sequences resident in memory including the ones being prefilled. `ok` names the first limit hit: `memory`, `max_seqs`, or `token_budget`. Every number is a `[low~high]` range around a mid value.
58
+
59
+ ### Install
60
+
61
+ | Command | What you get |
62
+ |---|---|
63
+ | `pip install servcalc` | Estimator and CLI. Depends only on `vramcalc`. Takes a local `config.json` or a model directory |
64
+ | `pip install "servcalc[hf]"` | Hugging Face Hub repo ids as input (`huggingface_hub`) |
65
+
66
+ Offline or on-premises: pass a model directory. With `HF_HUB_OFFLINE=1`, Hub ids resolve from the local cache only.
67
+
68
+ ### Why
69
+
70
+ Serving questions arrive in the same breath as memory questions: "Will an 8B model on a 3090 handle twenty users at a readable speed?" The usual answer is to rent the GPU, start vLLM and run a load test. That works, but it costs an afternoon per configuration, and the result says nothing about why the number is what it is. servcalc writes the roofline model down explicitly. Decode steps are bounded by memory bandwidth (weights and KV cache read once per step), prefill is bounded by compute, and an engine's scheduler decides how the two share a step. Each term is a formula you can read in `docs/design.md`, so when a measurement disagrees you know which assumption to fix.
71
+
72
+ ### Scope
73
+
74
+ - Models: dense decoder-only models that vramcalc parses (`llama`, `mistral`, `qwen2`, `gemma`, `gemma2`). GQA, tied embeddings and QKV bias are read from the config.
75
+ - GPUs: 34 cards from RTX 3060 to B200, with memory bandwidth and dense bf16 TFLOPs from vendor spec sheets. Pass a `GpuSpec` for anything else.
76
+ - Engines, as value-only profiles: `vllm` 0.6.3 (continuous batching, chunked prefill with a 512-token budget, paged KV in blocks of 16, `max_num_seqs` 256), `ollama` 0.5.x (llama.cpp server, GGUF Q4_K_M by default, `num_parallel` 4, `num_ctx` 2048 preallocated per slot), `hf` (`transformers` `generate()` in bf16, static batching, KV cache reallocated every step).
77
+ - Workload: equal prompt and generation lengths for every request, a list of batch sizes to sweep.
78
+
79
+ Out of scope for 0.1: MoE, MLA and sliding-window attention, speculative decoding, prefix caching, multi-GPU (tensor or pipeline parallel), CPU offload, admission queueing, and request length distributions. These are all deliberate: each one is a separate term or a separate model, and 0.1 keeps the closed-form core small enough to validate.
80
+
81
+ ### How it is computed
82
+
83
+ **Work.** For a decode step with `N` sequences and total context `sum_ctx`, bytes moved are `W_stream + KV_tok x sum_ctx + KV_tok x N`, where `W_stream` is the weight bytes read per step (input embedding rows are a lookup and excluded) and `KV_tok` the KV bytes per token. FLOPs are `2 x P_mm x N + 4 x L x q x sum_ctx`. A prefill chunk of `C` tokens with `c0` tokens already cached costs `2 x P_body x C + 4 x L x q x (C x c0 + C(C+1)/2)` FLOPs and writes `KV_tok x C` bytes. Summing chunks reproduces the whole-prompt cost exactly, whatever the chunk size.
84
+
85
+ **Time.** `step_ms = max(bytes / (bandwidth x bw_util), flops / (tflops x mfu)) + overhead`. `bw_util` and `mfu` are efficiency curves over the number of tokens in the step; `overhead` is the per-step scheduler and launch cost. Each curve has a low, mid and high value, which is where the ranges come from. The step is labelled `memory`, `compute` or `overhead` by its largest term at mid, and `saturating batch` is the first batch where decode flips from memory- to compute-bound.
86
+
87
+ **Modes.** Static batching (hf) runs one prefill for the whole batch, then `gen_len - 1` decode steps whose context grows each step; ITL is their mean. Continuous batching (vllm, ollama) is solved in steady state: with `N` sequences each producing `gen_len - 1` tokens, requests arrive at `N / (gen_len - 1)` per step. When the engine mixes prefill into decode steps (vLLM chunked prefill), the prompt is split into `k = ceil(prompt_len / (budget - N))` chunks and TTFT is the sum of those mixed steps; ITL is the frequency-weighted mean of mixed and decode-only steps. When prefill runs as its own pass (llama.cpp), ITL is `s_decode + arrival_rate x t_prefill` and TTFT is `t_prefill + s_decode / 2`.
88
+
89
+ **Profiles and capacity.** A profile is a dataclass of values: batching mode, sequence cap, quantization, KV bits and block size, memory budget and workspace ratio, token budget, whether prefill is mixed, whether KV is reallocated. Capacity is `(VRAM x budget - weights - VRAM x workspace) / KV per sequence at peak length`, rounded to KV blocks or to `num_ctx` for ollama. A batch is infeasible when its resident sequences exceed memory, the sequence cap, or the token budget, in that order.
90
+
91
+ ### Accuracy
92
+
93
+ **All numbers in 0.1 are prior ranges.** The efficiency curves are informed guesses from published vLLM and llama.cpp benchmarks, not measurements, and `calibration: prior` in the output says so. servcalc does not claim an error figure yet. 0.2 will run vLLM's `benchmark_serving.py` and llama.cpp's server benchmark on rented GPUs, fit the curves per engine and quantization, and report the residual the same way vramcalc reports its 19-run table. Until then, read the ranges as "the model is somewhere in here if the formulas hold", and use the `bound` column and `saturating batch` as the qualitative result.
94
+
95
+ ### Limits
96
+
97
+ - The steady-state solution is a fluid approximation: it assumes arrivals spread evenly across steps. Real schedulers admit requests in bursts, which widens TTFT.
98
+ - Quantized weight bytes count non-linear tensors (embeddings, norms) as 16-bit. Pass `weight_bytes=` with the real checkpoint size if you have it.
99
+ - `bf16_tflops` for T4 and V100 is the fp16 tensor-core peak since those cards have no bf16.
100
+ - Whether ollama's llama.cpp build mixes prefill into decode steps depends on its version and flags. 0.1 models it as a separate pass; 0.2 will confirm.
101
+
102
+ ### License
103
+
104
+ MIT
105
+
106
+ ---
107
+
108
+ ## 한국어
109
+
110
+ GPU를 빌리기 전에, 하나의 GPU가 LLM 사용자를 동시에 몇 명까지 어느 속도로 서빙할 수 있는지 예측하는 라이브러리입니다. Hugging Face `config.json`, GPU 이름, 서빙 엔진(vLLM, ollama, 일반 `transformers`)을 입력받아 배치 크기별로 상주 시퀀스 수, 첫 토큰 시간(TTFT), 토큰 간 지연(ITL), 초당 토큰 수, 그리고 해당 배치가 선택한 엔진에서 실행 가능한지를 반환합니다. [vramcalc](https://github.com/sacom123/vramcalc)의 다음 질문에 해당합니다. vramcalc는 "들어가는가"에, servcalc는 "얼마나 빠르게, 몇 명에게"에 답합니다.
111
+
112
+ ```python
113
+ from servcalc import Workload, estimate
114
+
115
+ e = estimate("meta-llama/Llama-3.1-8B", "RTX 3090", workload=Workload(prompt_len=1024, gen_len=512))
116
+ print(e.table())
117
+ print(e.max_users(itl_ms=50)) # 평균 ITL이 50 ms 이내인 가장 큰 실행 가능 배치
118
+ for p in e.points:
119
+ print(p.batch, p.feasible, p.itl_ms.mid, p.total_tok_s.mid)
120
+ ```
121
+
122
+ ```
123
+ $ servcalc meta-llama/Llama-3.1-8B --gpu "RTX 3090" --concurrency 1,4,16,64 --slo-itl 50
124
+ Llama-3.1-8B on RTX 3090 via vllm (0.6.3)
125
+ weights 16.06 GB resident, 15.01 GB streamed per step; KV 131.1 KB/token
126
+ capacity: memory 29 seqs, configured 29 seqs; saturating batch 16
127
+ calibration: prior (ranges are low~high)
128
+
129
+ N R ok TTFT ms ITL ms seq tok/s total tok/s bound
130
+ 1 2 yes 579 [482~724] 29.7 [24.7~36.2] 33.7 [27.6~40.5] 34 [28~40] memory
131
+ 4 5 yes 600 [493~768] 30.8 [25.8~43.5] 32.5 [23.0~38.8] 130 [92~155] memory
132
+ 16 17 yes 638 [522~821] 68.6 [50.8~105.2] 14.6 [9.5~19.7] 233 [152~315] compute
133
+ 64 65 no:memory 747 [612~956] 139.9 [109.6~194.3] 7.1 [5.1~9.1] 457 [329~584] compute
134
+
135
+ max users with mean ITL <= 50 ms: 4
136
+ ```
137
+
138
+ `N`은 decode 중인 시퀀스 수, `R`은 prefill 중인 것까지 포함해 메모리에 상주하는 시퀀스 수입니다. `ok`는 처음 걸리는 한계를 표시합니다(`memory`, `max_seqs`, `token_budget`). 모든 수치는 중앙값과 `[low~high]` 범위로 제공됩니다.
139
+
140
+ ### 설치
141
+
142
+ | 명령 | 제공 내용 |
143
+ |---|---|
144
+ | `pip install servcalc` | 추정기와 CLI. 의존성은 `vramcalc` 하나입니다. 로컬 `config.json` 또는 모델 디렉터리를 입력받습니다 |
145
+ | `pip install "servcalc[hf]"` | Hugging Face Hub 저장소 id 입력(`huggingface_hub`) |
146
+
147
+ 오프라인 또는 온프레미스 환경에서는 모델 디렉터리를 전달하면 됩니다. `HF_HUB_OFFLINE=1`을 설정하면 Hub id는 로컬 캐시에서만 해석됩니다.
148
+
149
+ ### 만든 이유
150
+
151
+ 서빙에 관한 질문은 메모리 질문과 함께 옵니다. "3090에서 8B 모델로 사용자 20명을 읽을 만한 속도로 받을 수 있는가"가 대표적입니다. 통상적인 답은 GPU를 빌려 vLLM을 올리고 부하 시험을 돌리는 것입니다. 유효한 방법이지만 구성 하나에 반나절이 들고, 결과 수치가 왜 그 값인지는 알려 주지 않습니다. servcalc는 roofline 모델을 명시적으로 기술합니다. decode step은 메모리 대역폭(매 step 가중치와 KV 캐시를 한 번 읽음)에, prefill은 연산량에 묶이며, 엔진의 스케줄러가 둘을 한 step에서 어떻게 섞는지를 결정합니다. 각 항은 `docs/design.md`에 식으로 적혀 있으므로, 실측과 다를 때 어느 가정을 고쳐야 하는지 알 수 있습니다.
152
+
153
+ ### 지원 범위
154
+
155
+ - 모델: vramcalc가 해석하는 dense decoder-only 모델(`llama`, `mistral`, `qwen2`, `gemma`, `gemma2`). GQA, tied embedding, QKV bias는 config에서 읽습니다.
156
+ - GPU: RTX 3060부터 B200까지 34종. 메모리 대역폭과 dense bf16 TFLOPs는 제조사 사양표 기준입니다. 그 외 장비는 `GpuSpec`으로 전달합니다.
157
+ - 엔진(값으로만 구성된 프로필): `vllm` 0.6.3(continuous batching, 512 토큰 예산의 chunked prefill, 16 토큰 블록 단위 paged KV, `max_num_seqs` 256), `ollama` 0.5.x(llama.cpp server, 기본 GGUF Q4_K_M, `num_parallel` 4, 슬롯당 `num_ctx` 2048 선할당), `hf`(`transformers` `generate()` bf16, static batching, 매 step KV 캐시 재할당).
158
+ - 워크로드: 모든 요청의 prompt 길이와 생성 길이가 같다고 가정하며, 배치 크기 목록을 순회합니다.
159
+
160
+ 0.1에서 제외한 항목: MoE, MLA, sliding-window attention, speculative decoding, prefix caching, 다중 GPU(tensor 또는 pipeline parallel), CPU offload, 대기열, 요청 길이 분포. 각각 별도의 항 또는 별도의 모델이 필요하므로, 0.1은 검증 가능한 크기의 닫힌 식 핵심만 담았습니다.
161
+
162
+ ### 계산 구조
163
+
164
+ **작업량.** 시퀀스 `N`개, 총 컨텍스트 `sum_ctx`인 decode step의 이동 바이트는 `W_stream + KV_tok x sum_ctx + KV_tok x N`입니다. `W_stream`은 step마다 읽는 가중치 바이트(입력 embedding은 조회이므로 제외), `KV_tok`은 토큰당 KV 바이트입니다. FLOPs는 `2 x P_mm x N + 4 x L x q x sum_ctx`입니다. 이미 `c0` 토큰이 캐시된 상태에서 `C` 토큰을 prefill하는 조각은 `2 x P_body x C + 4 x L x q x (C x c0 + C(C+1)/2)` FLOPs를 소비하고 `KV_tok x C` 바이트를 기록합니다. 조각의 합은 조각 크기와 무관하게 전체 prompt 비용과 정확히 일치합니다.
165
+
166
+ **시간.** `step_ms = max(bytes / (bandwidth x bw_util), flops / (tflops x mfu)) + overhead`입니다. `bw_util`과 `mfu`는 step 내 토큰 수에 대한 효율 곡선이고, `overhead`는 step당 스케줄러와 커널 실행 비용입니다. 각 곡선이 low, mid, high 값을 가지므로 결과가 범위로 나옵니다. step은 mid 기준으로 가장 큰 항에 따라 `memory`, `compute`, `overhead`로 표시되고, `saturating batch`는 decode가 memory-bound에서 compute-bound로 바뀌는 첫 배치입니다.
167
+
168
+ **모드.** static batching(hf)은 배치 전체를 한 번 prefill한 뒤 컨텍스트가 매 step 늘어나는 `gen_len - 1`회의 decode step을 수행하며, ITL은 그 평균입니다. continuous batching(vllm, ollama)은 정상 상태로 풉니다. 시퀀스 `N`개가 각각 `gen_len - 1` 토큰을 생성하므로 요청은 step당 `N / (gen_len - 1)`개 도착합니다. 엔진이 prefill을 decode step에 섞는 경우(vLLM chunked prefill) prompt는 `k = ceil(prompt_len / (budget - N))`개 조각으로 나뉘고 TTFT는 그 혼합 step들의 합, ITL은 혼합 step과 decode 전용 step의 빈도 가중 평균입니다. prefill이 별도 pass로 실행되는 경우(llama.cpp) ITL은 `s_decode + arrival_rate x t_prefill`, TTFT는 `t_prefill + s_decode / 2`입니다.
169
+
170
+ **프로필과 용량.** 프로필은 값만 담은 dataclass입니다. batching 방식, 시퀀스 상한, 양자화, KV 비트와 블록 크기, 메모리 예산과 workspace 비율, 토큰 예산, prefill 혼합 여부, KV 재할당 여부가 들어갑니다. 용량은 `(VRAM x budget - weights - VRAM x workspace) / 최대 길이 시 시퀀스당 KV`이며, KV 블록 또는 ollama의 `num_ctx` 단위로 올림합니다. 상주 시퀀스가 메모리, 시퀀스 상한, 토큰 예산을 이 순서로 초과하면 해당 배치는 실행 불가로 표시됩니다.
171
+
172
+ ### 정확도
173
+
174
+ **0.1의 모든 수치는 실측 전 사전 범위입니다.** 효율 곡선은 공개된 vLLM과 llama.cpp 벤치마크를 참고한 추정값이며 측정값이 아닙니다. 출력의 `calibration: prior`가 이를 표시합니다. servcalc는 아직 오차 수치를 주장하지 않습니다. 0.2에서는 임대 GPU에서 vLLM의 `benchmark_serving.py`와 llama.cpp server 벤치마크를 실행해 엔진과 양자화별로 곡선을 맞추고, vramcalc의 19회 실측 표와 같은 방식으로 잔차를 보고할 예정입니다. 그때까지 범위는 "식이 맞다면 이 안에 있다"로 읽고, `bound` 열과 `saturating batch`를 정성적 결과로 활용하시기 바랍니다.
175
+
176
+ ### 한계
177
+
178
+ - 정상 상태 해는 유체 근사입니다. 요청 도착이 step에 균등하게 퍼진다고 가정하므로, 실제 스케줄러의 묶음 단위 admission은 TTFT를 더 넓게 만듭니다.
179
+ - 양자화 가중치 바이트는 비선형 텐서(embedding, norm)를 16비트로 계산합니다. 실제 체크포인트 크기를 알고 있다면 `weight_bytes=`로 전달하시기 바랍니다.
180
+ - T4와 V100은 bf16이 없으므로 `bf16_tflops`에 fp16 tensor core 피크를 사용합니다.
181
+ - ollama의 llama.cpp 빌드가 prefill을 decode step에 섞는지는 버전과 옵션에 따라 다릅니다. 0.1은 별도 pass로 모델링하며 0.2에서 확인합니다.
182
+
183
+ ### 라이선스
184
+
185
+ MIT
@@ -0,0 +1,16 @@
1
+ servcalc/__init__.py,sha256=hrZMCNNv7TK11KqQeIXPu_eNBm-ptMlEYdvBY9Rn-ww,601
2
+ servcalc/calibration.py,sha256=WlGjYudarBE-K0vyaeNUzfjJHLqbzP-6XsLOfmdzuJw,3473
3
+ servcalc/cli.py,sha256=Zn4cBHAHICwwsW-gpQ-9VdNbnAT7yvwF2Jo6i3u72Fk,4707
4
+ servcalc/estimate.py,sha256=XcDmR6pgdnjpaGlHDeIOs0URlGlRaNNg6dvIyx8kh1w,6207
5
+ servcalc/gpus.py,sha256=hjq5LON16LVqD7nPAe1TSIBlGmyOhgpGAvlJqmqFu0E,2358
6
+ servcalc/modes.py,sha256=AmDM44ueVMq0-TTPb9qDUzWI3VH-kDK7RE5vjVVmId8,6799
7
+ servcalc/profiles.py,sha256=if0cHU4XHFzHo-jy09_FDLxkfU4st6AUjWSVdB9ZMyc,6785
8
+ servcalc/quant.py,sha256=JOwO68FIj5doFVBKFL8z0M8NN4Q1OGbcJvGB_FMjKEQ,893
9
+ servcalc/timing.py,sha256=quyGRAIYjmLCXCJVLuWiZfqs9pyjvCUZDkKASLkPn3Q,2027
10
+ servcalc/work.py,sha256=jnrKiIKp3r9pjw1n1KhkpL9eAuC5B3R6Vh7TqJt-Sak,2804
11
+ servcalc/workload.py,sha256=u-MCiL4NgzqZddgP7QdiTWfok4Ko9eD6C-AMy-l4H0Y,1429
12
+ servcalc-0.1.0.dist-info/METADATA,sha256=Z5hF51l0VufucCpAlPIVjzi5I5Qltv4PsEUeLBGSvt0,17765
13
+ servcalc-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
14
+ servcalc-0.1.0.dist-info/entry_points.txt,sha256=1UN_Okjyezn5eLLvJ7YCQvZLqBREKYBXe5Pd8Bh6aQ4,47
15
+ servcalc-0.1.0.dist-info/licenses/LICENSE,sha256=sZhDRicBLDYEeQNMKQ5o6sRsujSsfddgm_NG8u0a5iY,1068
16
+ servcalc-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ servcalc = servcalc.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Seonho Hong
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.