landauer-gap 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- landauer_gap-0.4.0/LICENSE +21 -0
- landauer_gap-0.4.0/PKG-INFO +200 -0
- landauer_gap-0.4.0/README.md +175 -0
- landauer_gap-0.4.0/pyproject.toml +39 -0
- landauer_gap-0.4.0/setup.cfg +4 -0
- landauer_gap-0.4.0/src/joules/__init__.py +2 -0
- landauer_gap-0.4.0/src/joules/__main__.py +5 -0
- landauer_gap-0.4.0/src/joules/bench.py +58 -0
- landauer_gap-0.4.0/src/joules/ci.py +175 -0
- landauer_gap-0.4.0/src/joules/cli.py +631 -0
- landauer_gap-0.4.0/src/joules/meters.py +636 -0
- landauer_gap-0.4.0/src/joules/model.py +60 -0
- landauer_gap-0.4.0/src/joules/proxy.py +455 -0
- landauer_gap-0.4.0/src/joules/sampler.py +54 -0
- landauer_gap-0.4.0/src/joules/share.py +164 -0
- landauer_gap-0.4.0/src/landauer_gap.egg-info/PKG-INFO +200 -0
- landauer_gap-0.4.0/src/landauer_gap.egg-info/SOURCES.txt +19 -0
- landauer_gap-0.4.0/src/landauer_gap.egg-info/dependency_links.txt +1 -0
- landauer_gap-0.4.0/src/landauer_gap.egg-info/entry_points.txt +3 -0
- landauer_gap-0.4.0/src/landauer_gap.egg-info/top_level.txt +1 -0
- landauer_gap-0.4.0/tests/test_joules.py +813 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Bharat Sharma
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: landauer-gap
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Like `time`, but for energy: measure the GPU, CPU and DRAM energy, cost and carbon of any command or local model.
|
|
5
|
+
Author: Bharat Sharma
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://landauer-gap.vercel.app
|
|
8
|
+
Project-URL: Leaderboard, https://landauer-gap.vercel.app/#/board
|
|
9
|
+
Project-URL: Source, https://github.com/bsharma173860-oss/d3-research/tree/main/apps/landauer-gap/joules
|
|
10
|
+
Project-URL: Issues, https://github.com/bsharma173860-oss/d3-research/issues
|
|
11
|
+
Keywords: energy,gpu,nvml,rapl,carbon,llm,benchmark,landauer
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
14
|
+
Classifier: Operating System :: MacOS
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Intended Audience :: Developers
|
|
17
|
+
Classifier: Intended Audience :: Science/Research
|
|
18
|
+
Classifier: Environment :: Console
|
|
19
|
+
Classifier: Topic :: System :: Monitoring
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Dynamic: license-file
|
|
25
|
+
|
|
26
|
+
# joules
|
|
27
|
+
|
|
28
|
+
**Like `time`, but for energy.** Wrap any command and get the real GPU, CPU and DRAM energy it
|
|
29
|
+
used, what that cost, its carbon, and how far it sits above the Landauer limit of physics.
|
|
30
|
+
Benchmark a local model and get joules per token, and how that compares with an API's price.
|
|
31
|
+
|
|
32
|
+
Illustrative output:
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
$ joules -- python train.py
|
|
36
|
+
joules · python train.py · 2.31 h · exit 0
|
|
37
|
+
GPU 0 NVIDIA H100 80GB HBM3 4.91 MJ 590 W avg nvml counter
|
|
38
|
+
CPU package 0 612 kJ 74 W avg rapl counter
|
|
39
|
+
DRAM (package 0) 98.7 kJ 12 W avg rapl counter
|
|
40
|
+
──────────────────────────────────────────────────────────
|
|
41
|
+
total 5.62 MJ = 1.56 kWh
|
|
42
|
+
cost 0.156 at 0.1/kWh · CO₂ 577 g at 370 g/kWh
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
$ joules bench ollama:llama3.1:8b --api-price 0.20
|
|
47
|
+
per token 3.74 J · 1.04 kWh per 1M tokens
|
|
48
|
+
electricity 0.104 per 1M output tokens at 0.1/kWh
|
|
49
|
+
vs API 0.2 per 1M → electricity is 1.9× cheaper than the API (hardware cost not included)
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
No dependencies: only the Python standard library and your GPU driver.
|
|
53
|
+
|
|
54
|
+
## Install
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
pipx install landauer-gap # or: pip install landauer-gap
|
|
58
|
+
pipx install "git+https://github.com/bsharma173860-oss/d3-research#subdirectory=apps/landauer-gap/joules"
|
|
59
|
+
joules devices # what can be measured on this machine
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## Commands
|
|
63
|
+
|
|
64
|
+
| Command | What it does |
|
|
65
|
+
|---|---|
|
|
66
|
+
| `joules -- CMD …` / `joules run -- CMD …` | Measure a command. Its exit code passes through, so it works in CI. |
|
|
67
|
+
| `joules bench ollama:MODEL` | Energy per token of a model served by Ollama. |
|
|
68
|
+
| `joules bench openai:MODEL --url http://host:8000/v1` | Same for any OpenAI-compatible server: vLLM, llama.cpp, LM Studio, TGI. |
|
|
69
|
+
| `joules run --track PROJECT -- CMD …` | Measure and save the run to your private [Tracking page](https://landauer-gap.vercel.app/#/track). Also on `bench`. |
|
|
70
|
+
| `joules bench … --share` | Add the result to the public [leaderboard](https://landauer-gap.vercel.app/#/board). |
|
|
71
|
+
| `joules share FILE` | Share a result saved earlier with `--out FILE`. |
|
|
72
|
+
| `joules ci --budget 10 -- CMD …` | Energy check for a pull request: this change vs the base branch. |
|
|
73
|
+
| `joules proxy --upstream URL` | Energy receipt on every response of a model server. |
|
|
74
|
+
| `joules devices` | List the meters joules found, and why any are missing. |
|
|
75
|
+
|
|
76
|
+
Useful options: `--json` / `--out FILE` (machine-readable result), `--subtract-idle 5` (measure the
|
|
77
|
+
idle machine first and also report energy above idle), `--gpus 0,1`, `--price`, `--grid fr` or
|
|
78
|
+
`--ci 55`, `--pue 1.2`.
|
|
79
|
+
|
|
80
|
+
## Tracking your runs over time
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
export LANDAUER_API_KEY=lgk_…
|
|
84
|
+
joules run --track llama-ft --params 8 --tokens 2 -- python train.py
|
|
85
|
+
joules --track nightly -- ./train.sh # short form
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Each tracked run appears on the Tracking page within seconds: Landauer gap over time, energy per run
|
|
89
|
+
and totals for the project. joules sends the energy, time, cost, CO₂, hardware, operations and gap,
|
|
90
|
+
and a short label: the program and script name (`python train.py`) or your `--label`. It never sends
|
|
91
|
+
arguments, inline code, outputs or file contents. A tracking problem is reported but never changes
|
|
92
|
+
the command's exit code, so it is safe in CI.
|
|
93
|
+
|
|
94
|
+
## Energy check on every pull request
|
|
95
|
+
|
|
96
|
+
`joules ci` runs a command on your checkout and on the base branch (a temporary git worktree),
|
|
97
|
+
alternating runs so drift hits both sides, and compares the medians. It writes a Markdown summary,
|
|
98
|
+
can post one pull request comment that it updates on each push, and fails when energy grows by more
|
|
99
|
+
than `--budget` percent. A change smaller than the run-to-run spread is reported as noise, never as a
|
|
100
|
+
failure.
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
joules ci --base main --runs 3 --budget 10 -- pytest -q
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
In GitHub Actions use the [PR energy check action](../integrations/pr-energy-check/README.md). On
|
|
107
|
+
hosted runners, which have no energy counters, energy is estimated from CPU time and the comment says
|
|
108
|
+
so; on a self-hosted runner with a GPU or RAPL it is measured. `joules run --estimate` uses the same
|
|
109
|
+
estimate when a machine has no counters.
|
|
110
|
+
|
|
111
|
+
## Energy receipts for AI responses
|
|
112
|
+
|
|
113
|
+
`joules proxy` sits in front of a model server and measures the hardware while each request runs.
|
|
114
|
+
Point your client at the proxy instead of the server; nothing else changes.
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
joules proxy --upstream http://localhost:11434 # Ollama
|
|
118
|
+
joules proxy --upstream http://localhost:8000 --subtract-idle 5 # vLLM, measured above idle
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
```
|
|
122
|
+
$ curl -si localhost:8787/v1/chat/completions -d '{"model":"llama3.1:8b","messages":[…]}'
|
|
123
|
+
X-Energy-Joules: 41.2
|
|
124
|
+
X-Energy-Output-Tokens: 212
|
|
125
|
+
X-Energy-Joules-Per-Token: 0.194
|
|
126
|
+
X-Energy-CO2-Grams: 0.00423
|
|
127
|
+
X-Energy-Receipt: /joules/receipts/r-000017
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
- Streaming responses (SSE or Ollama's NDJSON) pass through untouched and carry `X-Energy-Receipt`;
|
|
131
|
+
the receipt at that address is complete when the stream ends.
|
|
132
|
+
- Tokens come from the response (`usage`, or Ollama's `eval_count`). Without them, streamed pieces
|
|
133
|
+
are counted and the receipt says `tokens_exact: false`.
|
|
134
|
+
- Meters measure whole devices, so in each sampling interval the energy is shared equally between
|
|
135
|
+
the requests running at that moment. With `--subtract-idle` receipts also carry energy above idle.
|
|
136
|
+
- `/joules/receipts` lists the newest receipts, `/joules/metrics` serves Prometheus metrics, and
|
|
137
|
+
`/joules/health` says what is measured.
|
|
138
|
+
- `--track PROJECT` sends one summary per model every 5 minutes to your Tracking page: never
|
|
139
|
+
prompts or outputs.
|
|
140
|
+
- It listens on 127.0.0.1 by default. `--host 0.0.0.0` exposes it, and your model server, to your
|
|
141
|
+
network.
|
|
142
|
+
|
|
143
|
+
## The leaderboard
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
export LANDAUER_API_KEY=lgk_… # free key: landauer-gap.vercel.app → API
|
|
147
|
+
joules bench ollama:llama3.1:8b --subtract-idle 5 --share
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
The leaderboard ranks models and machines by measured joules per output token, using the median of
|
|
151
|
+
everyone's runs. joules sends only the model name, hardware name and count, energy per token,
|
|
152
|
+
tokens per second, average power, output tokens, whether idle power was subtracted, which meters
|
|
153
|
+
were used and the joules version. It never sends prompts, outputs, host names or paths. Simulated
|
|
154
|
+
runs are refused, and you can remove your own entries on the Leaderboard page.
|
|
155
|
+
|
|
156
|
+
## Estimate vs measured, and calibration
|
|
157
|
+
|
|
158
|
+
Tell joules the size of the training job and it compares the measurement with the Landauer Gap
|
|
159
|
+
estimate, then works out what your hardware really achieved:
|
|
160
|
+
|
|
161
|
+
```
|
|
162
|
+
$ joules run --params 1 --tokens 2 -- python finetune.py # 1B model, 2B tokens, one H100
|
|
163
|
+
estimate vs measured 17 MJ estimated (40% MFU, 80% draw) → 14.8 MJ measured (-13%)
|
|
164
|
+
your hardware ran at 42.5% effective utilisation and 74% of rated power
|
|
165
|
+
open in the calculator with these numbers: https://landauer-gap.vercel.app/?s=…
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
The link opens the website calculator with your measured utilisation and power draw, so every
|
|
169
|
+
future estimate starts from your real hardware instead of a textbook default. If the job you
|
|
170
|
+
describe could not fit in the measured time, joules says so instead of reporting nonsense.
|
|
171
|
+
|
|
172
|
+
## What it measures, and how
|
|
173
|
+
|
|
174
|
+
| Hardware | Source | Kind |
|
|
175
|
+
|---|---|---|
|
|
176
|
+
| NVIDIA GPUs | NVML (driver library, via ctypes) | energy counter on Volta and newer; power sampling on older cards |
|
|
177
|
+
| AMD GPUs (Instinct, Radeon) | amdgpu hwmon (`energy1_input`, else `power1_average`) | counter or sampling |
|
|
178
|
+
| Intel GPUs (Arc, Flex, Max) | i915 / xe hwmon `energy1_input` | energy counter |
|
|
179
|
+
| Intel and AMD CPUs, DRAM | RAPL (`/sys/class/powercap`) | counter, wraparound handled |
|
|
180
|
+
| Apple Silicon Macs | `powermetrics` (needs sudo) | sampling: CPU + GPU + Neural Engine |
|
|
181
|
+
| Intel Macs | `powermetrics` (needs sudo) | sampling: CPU package (cores, integrated GPU, DRAM) |
|
|
182
|
+
|
|
183
|
+
Meters cover whole devices, so anything else running on the same GPU or CPU is included. Use
|
|
184
|
+
`--subtract-idle`, and keep the machine otherwise quiet, for the cleanest numbers. Power supply
|
|
185
|
+
losses, fans, networking and cooling are not measured; `--pue` adds a facility overhead.
|
|
186
|
+
|
|
187
|
+
`CUDA_VISIBLE_DEVICES` is respected, with indices, GPU UUIDs (as Kubernetes and Slurm set them) or MIG
|
|
188
|
+
slices (MIG measures the whole GPU). On AMD, `HIP_VISIBLE_DEVICES` / `ROCR_VISIBLE_DEVICES` and `--gpus` pick cards. Reading RAPL on recent Linux kernels needs root, or
|
|
189
|
+
`sudo chmod a+r /sys/class/powercap/intel-rapl:*/energy_uj`.
|
|
190
|
+
|
|
191
|
+
## Tests
|
|
192
|
+
|
|
193
|
+
```bash
|
|
194
|
+
PYTHONPATH=src python3 -m unittest discover -s tests -v
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
The real sysfs layouts are rebuilt in a temporary directory (including counter wraparound), the
|
|
198
|
+
NVML binding runs against a compiled stand-in for `libnvidia-ml.so.1`, and model servers are
|
|
199
|
+
local HTTP stand-ins. `JOULES_SIMULATE="gpu:H100:650,cpu:package 0:80"` runs everything on
|
|
200
|
+
constant simulated power for demos; results are always labelled SIMULATED.
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
# joules
|
|
2
|
+
|
|
3
|
+
**Like `time`, but for energy.** Wrap any command and get the real GPU, CPU and DRAM energy it
|
|
4
|
+
used, what that cost, its carbon, and how far it sits above the Landauer limit of physics.
|
|
5
|
+
Benchmark a local model and get joules per token, and how that compares with an API's price.
|
|
6
|
+
|
|
7
|
+
Illustrative output:
|
|
8
|
+
|
|
9
|
+
```
|
|
10
|
+
$ joules -- python train.py
|
|
11
|
+
joules · python train.py · 2.31 h · exit 0
|
|
12
|
+
GPU 0 NVIDIA H100 80GB HBM3 4.91 MJ 590 W avg nvml counter
|
|
13
|
+
CPU package 0 612 kJ 74 W avg rapl counter
|
|
14
|
+
DRAM (package 0) 98.7 kJ 12 W avg rapl counter
|
|
15
|
+
──────────────────────────────────────────────────────────
|
|
16
|
+
total 5.62 MJ = 1.56 kWh
|
|
17
|
+
cost 0.156 at 0.1/kWh · CO₂ 577 g at 370 g/kWh
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
```
|
|
21
|
+
$ joules bench ollama:llama3.1:8b --api-price 0.20
|
|
22
|
+
per token 3.74 J · 1.04 kWh per 1M tokens
|
|
23
|
+
electricity 0.104 per 1M output tokens at 0.1/kWh
|
|
24
|
+
vs API 0.2 per 1M → electricity is 1.9× cheaper than the API (hardware cost not included)
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
No dependencies: only the Python standard library and your GPU driver.
|
|
28
|
+
|
|
29
|
+
## Install
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
pipx install landauer-gap # or: pip install landauer-gap
|
|
33
|
+
pipx install "git+https://github.com/bsharma173860-oss/d3-research#subdirectory=apps/landauer-gap/joules"
|
|
34
|
+
joules devices # what can be measured on this machine
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Commands
|
|
38
|
+
|
|
39
|
+
| Command | What it does |
|
|
40
|
+
|---|---|
|
|
41
|
+
| `joules -- CMD …` / `joules run -- CMD …` | Measure a command. Its exit code passes through, so it works in CI. |
|
|
42
|
+
| `joules bench ollama:MODEL` | Energy per token of a model served by Ollama. |
|
|
43
|
+
| `joules bench openai:MODEL --url http://host:8000/v1` | Same for any OpenAI-compatible server: vLLM, llama.cpp, LM Studio, TGI. |
|
|
44
|
+
| `joules run --track PROJECT -- CMD …` | Measure and save the run to your private [Tracking page](https://landauer-gap.vercel.app/#/track). Also on `bench`. |
|
|
45
|
+
| `joules bench … --share` | Add the result to the public [leaderboard](https://landauer-gap.vercel.app/#/board). |
|
|
46
|
+
| `joules share FILE` | Share a result saved earlier with `--out FILE`. |
|
|
47
|
+
| `joules ci --budget 10 -- CMD …` | Energy check for a pull request: this change vs the base branch. |
|
|
48
|
+
| `joules proxy --upstream URL` | Energy receipt on every response of a model server. |
|
|
49
|
+
| `joules devices` | List the meters joules found, and why any are missing. |
|
|
50
|
+
|
|
51
|
+
Useful options: `--json` / `--out FILE` (machine-readable result), `--subtract-idle 5` (measure the
|
|
52
|
+
idle machine first and also report energy above idle), `--gpus 0,1`, `--price`, `--grid fr` or
|
|
53
|
+
`--ci 55`, `--pue 1.2`.
|
|
54
|
+
|
|
55
|
+
## Tracking your runs over time
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
export LANDAUER_API_KEY=lgk_…
|
|
59
|
+
joules run --track llama-ft --params 8 --tokens 2 -- python train.py
|
|
60
|
+
joules --track nightly -- ./train.sh # short form
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Each tracked run appears on the Tracking page within seconds: Landauer gap over time, energy per run
|
|
64
|
+
and totals for the project. joules sends the energy, time, cost, CO₂, hardware, operations and gap,
|
|
65
|
+
and a short label: the program and script name (`python train.py`) or your `--label`. It never sends
|
|
66
|
+
arguments, inline code, outputs or file contents. A tracking problem is reported but never changes
|
|
67
|
+
the command's exit code, so it is safe in CI.
|
|
68
|
+
|
|
69
|
+
## Energy check on every pull request
|
|
70
|
+
|
|
71
|
+
`joules ci` runs a command on your checkout and on the base branch (a temporary git worktree),
|
|
72
|
+
alternating runs so drift hits both sides, and compares the medians. It writes a Markdown summary,
|
|
73
|
+
can post one pull request comment that it updates on each push, and fails when energy grows by more
|
|
74
|
+
than `--budget` percent. A change smaller than the run-to-run spread is reported as noise, never as a
|
|
75
|
+
failure.
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
joules ci --base main --runs 3 --budget 10 -- pytest -q
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
In GitHub Actions use the [PR energy check action](../integrations/pr-energy-check/README.md). On
|
|
82
|
+
hosted runners, which have no energy counters, energy is estimated from CPU time and the comment says
|
|
83
|
+
so; on a self-hosted runner with a GPU or RAPL it is measured. `joules run --estimate` uses the same
|
|
84
|
+
estimate when a machine has no counters.
|
|
85
|
+
|
|
86
|
+
## Energy receipts for AI responses
|
|
87
|
+
|
|
88
|
+
`joules proxy` sits in front of a model server and measures the hardware while each request runs.
|
|
89
|
+
Point your client at the proxy instead of the server; nothing else changes.
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
joules proxy --upstream http://localhost:11434 # Ollama
|
|
93
|
+
joules proxy --upstream http://localhost:8000 --subtract-idle 5 # vLLM, measured above idle
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
```
|
|
97
|
+
$ curl -si localhost:8787/v1/chat/completions -d '{"model":"llama3.1:8b","messages":[…]}'
|
|
98
|
+
X-Energy-Joules: 41.2
|
|
99
|
+
X-Energy-Output-Tokens: 212
|
|
100
|
+
X-Energy-Joules-Per-Token: 0.194
|
|
101
|
+
X-Energy-CO2-Grams: 0.00423
|
|
102
|
+
X-Energy-Receipt: /joules/receipts/r-000017
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
- Streaming responses (SSE or Ollama's NDJSON) pass through untouched and carry `X-Energy-Receipt`;
|
|
106
|
+
the receipt at that address is complete when the stream ends.
|
|
107
|
+
- Tokens come from the response (`usage`, or Ollama's `eval_count`). Without them, streamed pieces
|
|
108
|
+
are counted and the receipt says `tokens_exact: false`.
|
|
109
|
+
- Meters measure whole devices, so in each sampling interval the energy is shared equally between
|
|
110
|
+
the requests running at that moment. With `--subtract-idle` receipts also carry energy above idle.
|
|
111
|
+
- `/joules/receipts` lists the newest receipts, `/joules/metrics` serves Prometheus metrics, and
|
|
112
|
+
`/joules/health` says what is measured.
|
|
113
|
+
- `--track PROJECT` sends one summary per model every 5 minutes to your Tracking page: never
|
|
114
|
+
prompts or outputs.
|
|
115
|
+
- It listens on 127.0.0.1 by default. `--host 0.0.0.0` exposes it, and your model server, to your
|
|
116
|
+
network.
|
|
117
|
+
|
|
118
|
+
## The leaderboard
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
export LANDAUER_API_KEY=lgk_… # free key: landauer-gap.vercel.app → API
|
|
122
|
+
joules bench ollama:llama3.1:8b --subtract-idle 5 --share
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
The leaderboard ranks models and machines by measured joules per output token, using the median of
|
|
126
|
+
everyone's runs. joules sends only the model name, hardware name and count, energy per token,
|
|
127
|
+
tokens per second, average power, output tokens, whether idle power was subtracted, which meters
|
|
128
|
+
were used and the joules version. It never sends prompts, outputs, host names or paths. Simulated
|
|
129
|
+
runs are refused, and you can remove your own entries on the Leaderboard page.
|
|
130
|
+
|
|
131
|
+
## Estimate vs measured, and calibration
|
|
132
|
+
|
|
133
|
+
Tell joules the size of the training job and it compares the measurement with the Landauer Gap
|
|
134
|
+
estimate, then works out what your hardware really achieved:
|
|
135
|
+
|
|
136
|
+
```
|
|
137
|
+
$ joules run --params 1 --tokens 2 -- python finetune.py # 1B model, 2B tokens, one H100
|
|
138
|
+
estimate vs measured 17 MJ estimated (40% MFU, 80% draw) → 14.8 MJ measured (-13%)
|
|
139
|
+
your hardware ran at 42.5% effective utilisation and 74% of rated power
|
|
140
|
+
open in the calculator with these numbers: https://landauer-gap.vercel.app/?s=…
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
The link opens the website calculator with your measured utilisation and power draw, so every
|
|
144
|
+
future estimate starts from your real hardware instead of a textbook default. If the job you
|
|
145
|
+
describe could not fit in the measured time, joules says so instead of reporting nonsense.
|
|
146
|
+
|
|
147
|
+
## What it measures, and how
|
|
148
|
+
|
|
149
|
+
| Hardware | Source | Kind |
|
|
150
|
+
|---|---|---|
|
|
151
|
+
| NVIDIA GPUs | NVML (driver library, via ctypes) | energy counter on Volta and newer; power sampling on older cards |
|
|
152
|
+
| AMD GPUs (Instinct, Radeon) | amdgpu hwmon (`energy1_input`, else `power1_average`) | counter or sampling |
|
|
153
|
+
| Intel GPUs (Arc, Flex, Max) | i915 / xe hwmon `energy1_input` | energy counter |
|
|
154
|
+
| Intel and AMD CPUs, DRAM | RAPL (`/sys/class/powercap`) | counter, wraparound handled |
|
|
155
|
+
| Apple Silicon Macs | `powermetrics` (needs sudo) | sampling: CPU + GPU + Neural Engine |
|
|
156
|
+
| Intel Macs | `powermetrics` (needs sudo) | sampling: CPU package (cores, integrated GPU, DRAM) |
|
|
157
|
+
|
|
158
|
+
Meters cover whole devices, so anything else running on the same GPU or CPU is included. Use
|
|
159
|
+
`--subtract-idle`, and keep the machine otherwise quiet, for the cleanest numbers. Power supply
|
|
160
|
+
losses, fans, networking and cooling are not measured; `--pue` adds a facility overhead.
|
|
161
|
+
|
|
162
|
+
`CUDA_VISIBLE_DEVICES` is respected, with indices, GPU UUIDs (as Kubernetes and Slurm set them) or MIG
|
|
163
|
+
slices (MIG measures the whole GPU). On AMD, `HIP_VISIBLE_DEVICES` / `ROCR_VISIBLE_DEVICES` and `--gpus` pick cards. Reading RAPL on recent Linux kernels needs root, or
|
|
164
|
+
`sudo chmod a+r /sys/class/powercap/intel-rapl:*/energy_uj`.
|
|
165
|
+
|
|
166
|
+
## Tests
|
|
167
|
+
|
|
168
|
+
```bash
|
|
169
|
+
PYTHONPATH=src python3 -m unittest discover -s tests -v
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
The real sysfs layouts are rebuilt in a temporary directory (including counter wraparound), the
|
|
173
|
+
NVML binding runs against a compiled stand-in for `libnvidia-ml.so.1`, and model servers are
|
|
174
|
+
local HTTP stand-ins. `JOULES_SIMULATE="gpu:H100:650,cpu:package 0:80"` runs everything on
|
|
175
|
+
constant simulated power for demos; results are always labelled SIMULATED.
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "landauer-gap"
|
|
7
|
+
version = "0.4.0"
|
|
8
|
+
description = "Like `time`, but for energy: measure the GPU, CPU and DRAM energy, cost and carbon of any command or local model."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Bharat Sharma" }]
|
|
14
|
+
keywords = ["energy", "gpu", "nvml", "rapl", "carbon", "llm", "benchmark", "landauer"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 4 - Beta",
|
|
17
|
+
"Operating System :: POSIX :: Linux",
|
|
18
|
+
"Operating System :: MacOS",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Intended Audience :: Developers",
|
|
21
|
+
"Intended Audience :: Science/Research",
|
|
22
|
+
"Environment :: Console",
|
|
23
|
+
"Topic :: System :: Monitoring",
|
|
24
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
25
|
+
]
|
|
26
|
+
dependencies = []
|
|
27
|
+
|
|
28
|
+
[project.urls]
|
|
29
|
+
Homepage = "https://landauer-gap.vercel.app"
|
|
30
|
+
Leaderboard = "https://landauer-gap.vercel.app/#/board"
|
|
31
|
+
Source = "https://github.com/bsharma173860-oss/d3-research/tree/main/apps/landauer-gap/joules"
|
|
32
|
+
Issues = "https://github.com/bsharma173860-oss/d3-research/issues"
|
|
33
|
+
|
|
34
|
+
[project.scripts]
|
|
35
|
+
joules = "joules.cli:main"
|
|
36
|
+
landauer-gap = "joules.cli:main"
|
|
37
|
+
|
|
38
|
+
[tool.setuptools.packages.find]
|
|
39
|
+
where = ["src"]
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Benchmarks a local model server and counts the tokens it produced.
|
|
2
|
+
|
|
3
|
+
Targets:
|
|
4
|
+
ollama:<model> Ollama on http://localhost:11434 (or --url)
|
|
5
|
+
openai:<model> any OpenAI-compatible server: vLLM, llama.cpp, LM Studio, TGI
|
|
6
|
+
(default --url http://localhost:8000/v1)
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import time
|
|
12
|
+
import urllib.error
|
|
13
|
+
import urllib.request
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
|
|
16
|
+
DEFAULT_PROMPT = ("Explain, in about 200 words, why data centres measure power usage effectiveness "
|
|
17
|
+
"and what a good value looks like.")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class Target:
|
|
22
|
+
kind: str # ollama | openai
|
|
23
|
+
model: str
|
|
24
|
+
url: str
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def parse_target(spec: str, url: str | None) -> Target:
|
|
28
|
+
kind, sep, model = spec.partition(":")
|
|
29
|
+
if not sep or kind not in ("ollama", "openai") or not model:
|
|
30
|
+
raise ValueError("target must be ollama:<model> or openai:<model>, e.g. ollama:llama3.1:8b")
|
|
31
|
+
default = "http://localhost:11434" if kind == "ollama" else "http://localhost:8000/v1"
|
|
32
|
+
return Target(kind, model, (url or default).rstrip("/"))
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _post(url: str, body: dict, timeout: float) -> dict:
|
|
36
|
+
req = urllib.request.Request(url, data=json.dumps(body).encode(), headers={"Content-Type": "application/json"})
|
|
37
|
+
try:
|
|
38
|
+
with urllib.request.urlopen(req, timeout=timeout) as r:
|
|
39
|
+
return json.loads(r.read().decode())
|
|
40
|
+
except urllib.error.HTTPError as e:
|
|
41
|
+
raise RuntimeError(f"{url} answered HTTP {e.code}: {e.read().decode(errors='replace')[:300]}") from None
|
|
42
|
+
except urllib.error.URLError as e:
|
|
43
|
+
raise RuntimeError(f"cannot reach {url} ({e.reason}). Is the model server running?") from None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def generate(t: Target, prompt: str, max_tokens: int, timeout: float = 600) -> dict:
|
|
47
|
+
"""One request. Returns output tokens, prompt tokens and the server-reported generation time."""
|
|
48
|
+
if t.kind == "ollama":
|
|
49
|
+
r = _post(f"{t.url}/api/generate", {"model": t.model, "prompt": prompt, "stream": False,
|
|
50
|
+
"options": {"num_predict": max_tokens, "temperature": 0}}, timeout)
|
|
51
|
+
return {"output_tokens": int(r.get("eval_count", 0)), "prompt_tokens": int(r.get("prompt_eval_count", 0)),
|
|
52
|
+
"gen_s": r.get("eval_duration", 0) / 1e9 or None}
|
|
53
|
+
t0 = time.monotonic()
|
|
54
|
+
r = _post(f"{t.url}/chat/completions", {"model": t.model, "messages": [{"role": "user", "content": prompt}],
|
|
55
|
+
"max_tokens": max_tokens, "temperature": 0}, timeout)
|
|
56
|
+
u = r.get("usage") or {}
|
|
57
|
+
return {"output_tokens": int(u.get("completion_tokens", 0)), "prompt_tokens": int(u.get("prompt_tokens", 0)),
|
|
58
|
+
"gen_s": time.monotonic() - t0}
|