landauer-gap 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Bharat Sharma
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,200 @@
1
+ Metadata-Version: 2.4
2
+ Name: landauer-gap
3
+ Version: 0.4.0
4
+ Summary: Like `time`, but for energy: measure the GPU, CPU and DRAM energy, cost and carbon of any command or local model.
5
+ Author: Bharat Sharma
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://landauer-gap.vercel.app
8
+ Project-URL: Leaderboard, https://landauer-gap.vercel.app/#/board
9
+ Project-URL: Source, https://github.com/bsharma173860-oss/d3-research/tree/main/apps/landauer-gap/joules
10
+ Project-URL: Issues, https://github.com/bsharma173860-oss/d3-research/issues
11
+ Keywords: energy,gpu,nvml,rapl,carbon,llm,benchmark,landauer
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Operating System :: POSIX :: Linux
14
+ Classifier: Operating System :: MacOS
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Intended Audience :: Developers
17
+ Classifier: Intended Audience :: Science/Research
18
+ Classifier: Environment :: Console
19
+ Classifier: Topic :: System :: Monitoring
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Requires-Python: >=3.9
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Dynamic: license-file
25
+
26
+ # joules
27
+
28
+ **Like `time`, but for energy.** Wrap any command and get the real GPU, CPU and DRAM energy it
29
+ used, what that cost, its carbon, and how far it sits above the Landauer limit of physics.
30
+ Benchmark a local model and get joules per token, and how that compares with an API's price.
31
+
32
+ Illustrative output:
33
+
34
+ ```
35
+ $ joules -- python train.py
36
+ joules · python train.py · 2.31 h · exit 0
37
+ GPU 0 NVIDIA H100 80GB HBM3 4.91 MJ 590 W avg nvml counter
38
+ CPU package 0 612 kJ 74 W avg rapl counter
39
+ DRAM (package 0) 98.7 kJ 12 W avg rapl counter
40
+ ──────────────────────────────────────────────────────────
41
+ total 5.62 MJ = 1.56 kWh
42
+ cost 0.156 at 0.1/kWh · CO₂ 577 g at 370 g/kWh
43
+ ```
44
+
45
+ ```
46
+ $ joules bench ollama:llama3.1:8b --api-price 0.20
47
+ per token 3.74 J · 1.04 kWh per 1M tokens
48
+ electricity 0.104 per 1M output tokens at 0.1/kWh
49
+ vs API 0.2 per 1M → electricity is 1.9× cheaper than the API (hardware cost not included)
50
+ ```
51
+
52
+ No dependencies: only the Python standard library and your GPU driver.
53
+
54
+ ## Install
55
+
56
+ ```bash
57
+ pipx install landauer-gap # or: pip install landauer-gap
58
+ pipx install "git+https://github.com/bsharma173860-oss/d3-research#subdirectory=apps/landauer-gap/joules"
59
+ joules devices # what can be measured on this machine
60
+ ```
61
+
62
+ ## Commands
63
+
64
+ | Command | What it does |
65
+ |---|---|
66
+ | `joules -- CMD …` / `joules run -- CMD …` | Measure a command. Its exit code passes through, so it works in CI. |
67
+ | `joules bench ollama:MODEL` | Energy per token of a model served by Ollama. |
68
+ | `joules bench openai:MODEL --url http://host:8000/v1` | Same for any OpenAI-compatible server: vLLM, llama.cpp, LM Studio, TGI. |
69
+ | `joules run --track PROJECT -- CMD …` | Measure and save the run to your private [Tracking page](https://landauer-gap.vercel.app/#/track). Also on `bench`. |
70
+ | `joules bench … --share` | Add the result to the public [leaderboard](https://landauer-gap.vercel.app/#/board). |
71
+ | `joules share FILE` | Share a result saved earlier with `--out FILE`. |
72
+ | `joules ci --budget 10 -- CMD …` | Energy check for a pull request: this change vs the base branch. |
73
+ | `joules proxy --upstream URL` | Energy receipt on every response of a model server. |
74
+ | `joules devices` | List the meters joules found, and why any are missing. |
75
+
76
+ Useful options: `--json` / `--out FILE` (machine-readable result), `--subtract-idle 5` (measure the
77
+ idle machine first and also report energy above idle), `--gpus 0,1`, `--price`, `--grid fr` or
78
+ `--ci 55`, `--pue 1.2`.
79
+
80
+ ## Tracking your runs over time
81
+
82
+ ```bash
83
+ export LANDAUER_API_KEY=lgk_…
84
+ joules run --track llama-ft --params 8 --tokens 2 -- python train.py
85
+ joules --track nightly -- ./train.sh # short form
86
+ ```
87
+
88
+ Each tracked run appears on the Tracking page within seconds: Landauer gap over time, energy per run
89
+ and totals for the project. joules sends the energy, time, cost, CO₂, hardware, operations and gap,
90
+ and a short label: the program and script name (`python train.py`) or your `--label`. It never sends
91
+ arguments, inline code, outputs or file contents. A tracking problem is reported but never changes
92
+ the command's exit code, so it is safe in CI.
93
+
94
+ ## Energy check on every pull request
95
+
96
+ `joules ci` runs a command on your checkout and on the base branch (a temporary git worktree),
97
+ alternating runs so drift hits both sides, and compares the medians. It writes a Markdown summary,
98
+ can post one pull request comment that it updates on each push, and fails when energy grows by more
99
+ than `--budget` percent. A change smaller than the run-to-run spread is reported as noise, never as a
100
+ failure.
101
+
102
+ ```bash
103
+ joules ci --base main --runs 3 --budget 10 -- pytest -q
104
+ ```
105
+
106
+ In GitHub Actions use the [PR energy check action](../integrations/pr-energy-check/README.md). On
107
+ hosted runners, which have no energy counters, energy is estimated from CPU time and the comment says
108
+ so; on a self-hosted runner with a GPU or RAPL it is measured. `joules run --estimate` uses the same
109
+ estimate when a machine has no counters.
110
+
111
+ ## Energy receipts for AI responses
112
+
113
+ `joules proxy` sits in front of a model server and measures the hardware while each request runs.
114
+ Point your client at the proxy instead of the server; nothing else changes.
115
+
116
+ ```bash
117
+ joules proxy --upstream http://localhost:11434 # Ollama
118
+ joules proxy --upstream http://localhost:8000 --subtract-idle 5 # vLLM, measured above idle
119
+ ```
120
+
121
+ ```
122
+ $ curl -si localhost:8787/v1/chat/completions -d '{"model":"llama3.1:8b","messages":[…]}'
123
+ X-Energy-Joules: 41.2
124
+ X-Energy-Output-Tokens: 212
125
+ X-Energy-Joules-Per-Token: 0.194
126
+ X-Energy-CO2-Grams: 0.00423
127
+ X-Energy-Receipt: /joules/receipts/r-000017
128
+ ```
129
+
130
+ - Streaming responses (SSE or Ollama's NDJSON) pass through untouched and carry `X-Energy-Receipt`;
131
+ the receipt at that address is complete when the stream ends.
132
+ - Tokens come from the response (`usage`, or Ollama's `eval_count`). Without them, streamed pieces
133
+ are counted and the receipt says `tokens_exact: false`.
134
+ - Meters measure whole devices, so in each sampling interval the energy is shared equally between
135
+ the requests running at that moment. With `--subtract-idle` receipts also carry energy above idle.
136
+ - `/joules/receipts` lists the newest receipts, `/joules/metrics` serves Prometheus metrics, and
137
+ `/joules/health` says what is measured.
138
+ - `--track PROJECT` sends one summary per model every 5 minutes to your Tracking page: never
139
+ prompts or outputs.
140
+ - It listens on 127.0.0.1 by default. `--host 0.0.0.0` exposes it, and your model server, to your
141
+ network.
142
+
143
+ ## The leaderboard
144
+
145
+ ```bash
146
+ export LANDAUER_API_KEY=lgk_… # free key: landauer-gap.vercel.app → API
147
+ joules bench ollama:llama3.1:8b --subtract-idle 5 --share
148
+ ```
149
+
150
+ The leaderboard ranks models and machines by measured joules per output token, using the median of
151
+ everyone's runs. joules sends only the model name, hardware name and count, energy per token,
152
+ tokens per second, average power, output tokens, whether idle power was subtracted, which meters
153
+ were used and the joules version. It never sends prompts, outputs, host names or paths. Simulated
154
+ runs are refused, and you can remove your own entries on the Leaderboard page.
155
+
156
+ ## Estimate vs measured, and calibration
157
+
158
+ Tell joules the size of the training job and it compares the measurement with the Landauer Gap
159
+ estimate, then works out what your hardware really achieved:
160
+
161
+ ```
162
+ $ joules run --params 1 --tokens 2 -- python finetune.py # 1B model, 2B tokens, one H100
163
+ estimate vs measured 17 MJ estimated (40% MFU, 80% draw) → 14.8 MJ measured (-13%)
164
+ your hardware ran at 42.5% effective utilisation and 74% of rated power
165
+ open in the calculator with these numbers: https://landauer-gap.vercel.app/?s=…
166
+ ```
167
+
168
+ The link opens the website calculator with your measured utilisation and power draw, so every
169
+ future estimate starts from your real hardware instead of a textbook default. If the job you
170
+ describe could not fit in the measured time, joules says so instead of reporting nonsense.
171
+
172
+ ## What it measures, and how
173
+
174
+ | Hardware | Source | Kind |
175
+ |---|---|---|
176
+ | NVIDIA GPUs | NVML (driver library, via ctypes) | energy counter on Volta and newer; power sampling on older cards |
177
+ | AMD GPUs (Instinct, Radeon) | amdgpu hwmon (`energy1_input`, else `power1_average`) | counter or sampling |
178
+ | Intel GPUs (Arc, Flex, Max) | i915 / xe hwmon `energy1_input` | energy counter |
179
+ | Intel and AMD CPUs, DRAM | RAPL (`/sys/class/powercap`) | counter, wraparound handled |
180
+ | Apple Silicon Macs | `powermetrics` (needs sudo) | sampling: CPU + GPU + Neural Engine |
181
+ | Intel Macs | `powermetrics` (needs sudo) | sampling: CPU package (cores, integrated GPU, DRAM) |
182
+
183
+ Meters cover whole devices, so anything else running on the same GPU or CPU is included. Use
184
+ `--subtract-idle`, and keep the machine otherwise quiet, for the cleanest numbers. Power supply
185
+ losses, fans, networking and cooling are not measured; `--pue` adds a facility overhead.
186
+
187
+ `CUDA_VISIBLE_DEVICES` is respected, with indices, GPU UUIDs (as Kubernetes and Slurm set them) or MIG
188
+ slices (MIG measures the whole GPU). On AMD, `HIP_VISIBLE_DEVICES` / `ROCR_VISIBLE_DEVICES` and `--gpus` pick cards. Reading RAPL on recent Linux kernels needs root, or
189
+ `sudo chmod a+r /sys/class/powercap/intel-rapl:*/energy_uj`.
190
+
191
+ ## Tests
192
+
193
+ ```bash
194
+ PYTHONPATH=src python3 -m unittest discover -s tests -v
195
+ ```
196
+
197
+ The real sysfs layouts are rebuilt in a temporary directory (including counter wraparound), the
198
+ NVML binding runs against a compiled stand-in for `libnvidia-ml.so.1`, and model servers are
199
+ local HTTP stand-ins. `JOULES_SIMULATE="gpu:H100:650,cpu:package 0:80"` runs everything on
200
+ constant simulated power for demos; results are always labelled SIMULATED.
@@ -0,0 +1,175 @@
1
+ # joules
2
+
3
+ **Like `time`, but for energy.** Wrap any command and get the real GPU, CPU and DRAM energy it
4
+ used, what that cost, its carbon, and how far it sits above the Landauer limit of physics.
5
+ Benchmark a local model and get joules per token, and how that compares with an API's price.
6
+
7
+ Illustrative output:
8
+
9
+ ```
10
+ $ joules -- python train.py
11
+ joules · python train.py · 2.31 h · exit 0
12
+ GPU 0 NVIDIA H100 80GB HBM3 4.91 MJ 590 W avg nvml counter
13
+ CPU package 0 612 kJ 74 W avg rapl counter
14
+ DRAM (package 0) 98.7 kJ 12 W avg rapl counter
15
+ ──────────────────────────────────────────────────────────
16
+ total 5.62 MJ = 1.56 kWh
17
+ cost 0.156 at 0.1/kWh · CO₂ 577 g at 370 g/kWh
18
+ ```
19
+
20
+ ```
21
+ $ joules bench ollama:llama3.1:8b --api-price 0.20
22
+ per token 3.74 J · 1.04 kWh per 1M tokens
23
+ electricity 0.104 per 1M output tokens at 0.1/kWh
24
+ vs API 0.2 per 1M → electricity is 1.9× cheaper than the API (hardware cost not included)
25
+ ```
26
+
27
+ No dependencies: only the Python standard library and your GPU driver.
28
+
29
+ ## Install
30
+
31
+ ```bash
32
+ pipx install landauer-gap # or: pip install landauer-gap
33
+ pipx install "git+https://github.com/bsharma173860-oss/d3-research#subdirectory=apps/landauer-gap/joules"
34
+ joules devices # what can be measured on this machine
35
+ ```
36
+
37
+ ## Commands
38
+
39
+ | Command | What it does |
40
+ |---|---|
41
+ | `joules -- CMD …` / `joules run -- CMD …` | Measure a command. Its exit code passes through, so it works in CI. |
42
+ | `joules bench ollama:MODEL` | Energy per token of a model served by Ollama. |
43
+ | `joules bench openai:MODEL --url http://host:8000/v1` | Same for any OpenAI-compatible server: vLLM, llama.cpp, LM Studio, TGI. |
44
+ | `joules run --track PROJECT -- CMD …` | Measure and save the run to your private [Tracking page](https://landauer-gap.vercel.app/#/track). Also on `bench`. |
45
+ | `joules bench … --share` | Add the result to the public [leaderboard](https://landauer-gap.vercel.app/#/board). |
46
+ | `joules share FILE` | Share a result saved earlier with `--out FILE`. |
47
+ | `joules ci --budget 10 -- CMD …` | Energy check for a pull request: this change vs the base branch. |
48
+ | `joules proxy --upstream URL` | Energy receipt on every response of a model server. |
49
+ | `joules devices` | List the meters joules found, and why any are missing. |
50
+
51
+ Useful options: `--json` / `--out FILE` (machine-readable result), `--subtract-idle 5` (measure the
52
+ idle machine first and also report energy above idle), `--gpus 0,1`, `--price`, `--grid fr` or
53
+ `--ci 55`, `--pue 1.2`.
54
+
55
+ ## Tracking your runs over time
56
+
57
+ ```bash
58
+ export LANDAUER_API_KEY=lgk_…
59
+ joules run --track llama-ft --params 8 --tokens 2 -- python train.py
60
+ joules --track nightly -- ./train.sh # short form
61
+ ```
62
+
63
+ Each tracked run appears on the Tracking page within seconds: Landauer gap over time, energy per run
64
+ and totals for the project. joules sends the energy, time, cost, CO₂, hardware, operations and gap,
65
+ and a short label: the program and script name (`python train.py`) or your `--label`. It never sends
66
+ arguments, inline code, outputs or file contents. A tracking problem is reported but never changes
67
+ the command's exit code, so it is safe in CI.
68
+
69
+ ## Energy check on every pull request
70
+
71
+ `joules ci` runs a command on your checkout and on the base branch (a temporary git worktree),
72
+ alternating runs so drift hits both sides, and compares the medians. It writes a Markdown summary,
73
+ can post one pull request comment that it updates on each push, and fails when energy grows by more
74
+ than `--budget` percent. A change smaller than the run-to-run spread is reported as noise, never as a
75
+ failure.
76
+
77
+ ```bash
78
+ joules ci --base main --runs 3 --budget 10 -- pytest -q
79
+ ```
80
+
81
+ In GitHub Actions use the [PR energy check action](../integrations/pr-energy-check/README.md). On
82
+ hosted runners, which have no energy counters, energy is estimated from CPU time and the comment says
83
+ so; on a self-hosted runner with a GPU or RAPL it is measured. `joules run --estimate` uses the same
84
+ estimate when a machine has no counters.
85
+
86
+ ## Energy receipts for AI responses
87
+
88
+ `joules proxy` sits in front of a model server and measures the hardware while each request runs.
89
+ Point your client at the proxy instead of the server; nothing else changes.
90
+
91
+ ```bash
92
+ joules proxy --upstream http://localhost:11434 # Ollama
93
+ joules proxy --upstream http://localhost:8000 --subtract-idle 5 # vLLM, measured above idle
94
+ ```
95
+
96
+ ```
97
+ $ curl -si localhost:8787/v1/chat/completions -d '{"model":"llama3.1:8b","messages":[…]}'
98
+ X-Energy-Joules: 41.2
99
+ X-Energy-Output-Tokens: 212
100
+ X-Energy-Joules-Per-Token: 0.194
101
+ X-Energy-CO2-Grams: 0.00423
102
+ X-Energy-Receipt: /joules/receipts/r-000017
103
+ ```
104
+
105
+ - Streaming responses (SSE or Ollama's NDJSON) pass through untouched and carry `X-Energy-Receipt`;
106
+ the receipt at that address is complete when the stream ends.
107
+ - Tokens come from the response (`usage`, or Ollama's `eval_count`). Without them, streamed pieces
108
+ are counted and the receipt says `tokens_exact: false`.
109
+ - Meters measure whole devices, so in each sampling interval the energy is shared equally between
110
+ the requests running at that moment. With `--subtract-idle` receipts also carry energy above idle.
111
+ - `/joules/receipts` lists the newest receipts, `/joules/metrics` serves Prometheus metrics, and
112
+ `/joules/health` says what is measured.
113
+ - `--track PROJECT` sends one summary per model every 5 minutes to your Tracking page: never
114
+ prompts or outputs.
115
+ - It listens on 127.0.0.1 by default. `--host 0.0.0.0` exposes it, and your model server, to your
116
+ network.
117
+
118
+ ## The leaderboard
119
+
120
+ ```bash
121
+ export LANDAUER_API_KEY=lgk_… # free key: landauer-gap.vercel.app → API
122
+ joules bench ollama:llama3.1:8b --subtract-idle 5 --share
123
+ ```
124
+
125
+ The leaderboard ranks models and machines by measured joules per output token, using the median of
126
+ everyone's runs. joules sends only the model name, hardware name and count, energy per token,
127
+ tokens per second, average power, output tokens, whether idle power was subtracted, which meters
128
+ were used and the joules version. It never sends prompts, outputs, host names or paths. Simulated
129
+ runs are refused, and you can remove your own entries on the Leaderboard page.
130
+
131
+ ## Estimate vs measured, and calibration
132
+
133
+ Tell joules the size of the training job and it compares the measurement with the Landauer Gap
134
+ estimate, then works out what your hardware really achieved:
135
+
136
+ ```
137
+ $ joules run --params 1 --tokens 2 -- python finetune.py # 1B model, 2B tokens, one H100
138
+ estimate vs measured 17 MJ estimated (40% MFU, 80% draw) → 14.8 MJ measured (-13%)
139
+ your hardware ran at 42.5% effective utilisation and 74% of rated power
140
+ open in the calculator with these numbers: https://landauer-gap.vercel.app/?s=…
141
+ ```
142
+
143
+ The link opens the website calculator with your measured utilisation and power draw, so every
144
+ future estimate starts from your real hardware instead of a textbook default. If the job you
145
+ describe could not fit in the measured time, joules says so instead of reporting nonsense.
146
+
147
+ ## What it measures, and how
148
+
149
+ | Hardware | Source | Kind |
150
+ |---|---|---|
151
+ | NVIDIA GPUs | NVML (driver library, via ctypes) | energy counter on Volta and newer; power sampling on older cards |
152
+ | AMD GPUs (Instinct, Radeon) | amdgpu hwmon (`energy1_input`, else `power1_average`) | counter or sampling |
153
+ | Intel GPUs (Arc, Flex, Max) | i915 / xe hwmon `energy1_input` | energy counter |
154
+ | Intel and AMD CPUs, DRAM | RAPL (`/sys/class/powercap`) | counter, wraparound handled |
155
+ | Apple Silicon Macs | `powermetrics` (needs sudo) | sampling: CPU + GPU + Neural Engine |
156
+ | Intel Macs | `powermetrics` (needs sudo) | sampling: CPU package (cores, integrated GPU, DRAM) |
157
+
158
+ Meters cover whole devices, so anything else running on the same GPU or CPU is included. Use
159
+ `--subtract-idle`, and keep the machine otherwise quiet, for the cleanest numbers. Power supply
160
+ losses, fans, networking and cooling are not measured; `--pue` adds a facility overhead.
161
+
162
+ `CUDA_VISIBLE_DEVICES` is respected, with indices, GPU UUIDs (as Kubernetes and Slurm set them) or MIG
163
+ slices (MIG measures the whole GPU). On AMD, `HIP_VISIBLE_DEVICES` / `ROCR_VISIBLE_DEVICES` and `--gpus` pick cards. Reading RAPL on recent Linux kernels needs root, or
164
+ `sudo chmod a+r /sys/class/powercap/intel-rapl:*/energy_uj`.
165
+
166
+ ## Tests
167
+
168
+ ```bash
169
+ PYTHONPATH=src python3 -m unittest discover -s tests -v
170
+ ```
171
+
172
+ The real sysfs layouts are rebuilt in a temporary directory (including counter wraparound), the
173
+ NVML binding runs against a compiled stand-in for `libnvidia-ml.so.1`, and model servers are
174
+ local HTTP stand-ins. `JOULES_SIMULATE="gpu:H100:650,cpu:package 0:80"` runs everything on
175
+ constant simulated power for demos; results are always labelled SIMULATED.
@@ -0,0 +1,39 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "landauer-gap"
7
+ version = "0.4.0"
8
+ description = "Like `time`, but for energy: measure the GPU, CPU and DRAM energy, cost and carbon of any command or local model."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Bharat Sharma" }]
14
+ keywords = ["energy", "gpu", "nvml", "rapl", "carbon", "llm", "benchmark", "landauer"]
15
+ classifiers = [
16
+ "Development Status :: 4 - Beta",
17
+ "Operating System :: POSIX :: Linux",
18
+ "Operating System :: MacOS",
19
+ "Programming Language :: Python :: 3",
20
+ "Intended Audience :: Developers",
21
+ "Intended Audience :: Science/Research",
22
+ "Environment :: Console",
23
+ "Topic :: System :: Monitoring",
24
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
25
+ ]
26
+ dependencies = []
27
+
28
+ [project.urls]
29
+ Homepage = "https://landauer-gap.vercel.app"
30
+ Leaderboard = "https://landauer-gap.vercel.app/#/board"
31
+ Source = "https://github.com/bsharma173860-oss/d3-research/tree/main/apps/landauer-gap/joules"
32
+ Issues = "https://github.com/bsharma173860-oss/d3-research/issues"
33
+
34
+ [project.scripts]
35
+ joules = "joules.cli:main"
36
+ landauer-gap = "joules.cli:main"
37
+
38
+ [tool.setuptools.packages.find]
39
+ where = ["src"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,2 @@
1
+ """joules: measure the energy, cost and carbon of any command or local model."""
2
+ __version__ = "0.4.0"
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
@@ -0,0 +1,58 @@
1
+ """Benchmarks a local model server and counts the tokens it produced.
2
+
3
+ Targets:
4
+ ollama:<model> Ollama on http://localhost:11434 (or --url)
5
+ openai:<model> any OpenAI-compatible server: vLLM, llama.cpp, LM Studio, TGI
6
+ (default --url http://localhost:8000/v1)
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import time
12
+ import urllib.error
13
+ import urllib.request
14
+ from dataclasses import dataclass
15
+
16
+ DEFAULT_PROMPT = ("Explain, in about 200 words, why data centres measure power usage effectiveness "
17
+ "and what a good value looks like.")
18
+
19
+
20
+ @dataclass
21
+ class Target:
22
+ kind: str # ollama | openai
23
+ model: str
24
+ url: str
25
+
26
+
27
+ def parse_target(spec: str, url: str | None) -> Target:
28
+ kind, sep, model = spec.partition(":")
29
+ if not sep or kind not in ("ollama", "openai") or not model:
30
+ raise ValueError("target must be ollama:<model> or openai:<model>, e.g. ollama:llama3.1:8b")
31
+ default = "http://localhost:11434" if kind == "ollama" else "http://localhost:8000/v1"
32
+ return Target(kind, model, (url or default).rstrip("/"))
33
+
34
+
35
+ def _post(url: str, body: dict, timeout: float) -> dict:
36
+ req = urllib.request.Request(url, data=json.dumps(body).encode(), headers={"Content-Type": "application/json"})
37
+ try:
38
+ with urllib.request.urlopen(req, timeout=timeout) as r:
39
+ return json.loads(r.read().decode())
40
+ except urllib.error.HTTPError as e:
41
+ raise RuntimeError(f"{url} answered HTTP {e.code}: {e.read().decode(errors='replace')[:300]}") from None
42
+ except urllib.error.URLError as e:
43
+ raise RuntimeError(f"cannot reach {url} ({e.reason}). Is the model server running?") from None
44
+
45
+
46
+ def generate(t: Target, prompt: str, max_tokens: int, timeout: float = 600) -> dict:
47
+ """One request. Returns output tokens, prompt tokens and the server-reported generation time."""
48
+ if t.kind == "ollama":
49
+ r = _post(f"{t.url}/api/generate", {"model": t.model, "prompt": prompt, "stream": False,
50
+ "options": {"num_predict": max_tokens, "temperature": 0}}, timeout)
51
+ return {"output_tokens": int(r.get("eval_count", 0)), "prompt_tokens": int(r.get("prompt_eval_count", 0)),
52
+ "gen_s": r.get("eval_duration", 0) / 1e9 or None}
53
+ t0 = time.monotonic()
54
+ r = _post(f"{t.url}/chat/completions", {"model": t.model, "messages": [{"role": "user", "content": prompt}],
55
+ "max_tokens": max_tokens, "temperature": 0}, timeout)
56
+ u = r.get("usage") or {}
57
+ return {"output_tokens": int(u.get("completion_tokens", 0)), "prompt_tokens": int(u.get("prompt_tokens", 0)),
58
+ "gen_s": time.monotonic() - t0}