aimpg 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aimpg/__init__.py +1 -0
- aimpg/attribution.py +252 -0
- aimpg/cli.py +40 -0
- aimpg/energy.py +195 -0
- aimpg/factors.json +63 -0
- aimpg/gitkept.py +255 -0
- aimpg/logs.py +318 -0
- aimpg/model.py +97 -0
- aimpg/receipt.py +149 -0
- aimpg-0.1.0.dist-info/METADATA +93 -0
- aimpg-0.1.0.dist-info/RECORD +14 -0
- aimpg-0.1.0.dist-info/WHEEL +4 -0
- aimpg-0.1.0.dist-info/entry_points.txt +2 -0
- aimpg-0.1.0.dist-info/licenses/LICENSE +21 -0
aimpg/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""aimpg: real-world energy per solved task for AI coding agents."""
|
aimpg/attribution.py
ADDED
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
"""Tie AI requests to the commits they produced.
|
|
2
|
+
|
|
3
|
+
Tier 1 (exact): a Bash tool call ran `git commit`, and a commit in that
|
|
4
|
+
repo has a committer time inside the call's [start, end] interval (±2s
|
|
5
|
+
for whole-second git timestamps). Measured 8/8 on real commit history,
|
|
6
|
+
including a commit whose hooks made it land 147s after the call started.
|
|
7
|
+
|
|
8
|
+
Energy split (time segments): within one session, the requests made after
|
|
9
|
+
commit k-1 and up to commit k belong to commit k, whatever repo they ran
|
|
10
|
+
in. A session's cwd is often not the repo it works on (absolute paths,
|
|
11
|
+
`cd x && ...`), so grouping by folder would strand requests.
|
|
12
|
+
|
|
13
|
+
session s: r r r [c1 in A] r r r r [c2 in B] r r
|
|
14
|
+
└── c1 ───┘ └──── c2 ─────┘ └ no commit yet
|
|
15
|
+
|
|
16
|
+
Long sessions: a 2h+ pause starts a new work burst. Only the final burst
|
|
17
|
+
before a commit is its direct energy; earlier bursts since the previous
|
|
18
|
+
commit are reported as its "lead-up", never dropped.
|
|
19
|
+
|
|
20
|
+
Tier 2 (fuzzy): a session's leftover requests (after its last exact
|
|
21
|
+
commit) go to commits in any repo it worked in, made between its first leftover
|
|
22
|
+
request and 2h after its last, authored by you (repo user.email) that touch a file the session edited
|
|
23
|
+
(Edit/Write tools or Bash: sed -i, redirects, mv/cp/rm). Several matches
|
|
24
|
+
split the requests by lines changed. Commits after the session's last
|
|
25
|
+
request are labeled "grace". Built because the first real run showed exact
|
|
26
|
+
matches covered only ~48% of in-repo energy (design gate: 70%).
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import os
|
|
32
|
+
from collections import defaultdict
|
|
33
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
34
|
+
from dataclasses import dataclass, field
|
|
35
|
+
from typing import Callable
|
|
36
|
+
|
|
37
|
+
from aimpg.gitkept import DAY, Commit, GitError, KeptChecker, load_commits, repo_root, user_email
|
|
38
|
+
from aimpg.model import CommitCall, ParseResult, Request, Task
|
|
39
|
+
|
|
40
|
+
PAD = 2.0 # seconds; git timestamps are whole seconds
|
|
41
|
+
GRACE = 2 * 3600.0
|
|
42
|
+
BREAK = 2 * 3600.0 # a pause this long between requests starts a new work burst # hand commits often land a while after the session goes quiet
|
|
43
|
+
|
|
44
|
+
NOT_IN_REPO = "not in a git repo (or repo moved/deleted)"
|
|
45
|
+
NO_COMMIT_YET = "no commit from this session yet"
|
|
46
|
+
GIT_FAILED = "git error"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass
|
|
50
|
+
class RepoInfo:
|
|
51
|
+
ref: str = ""
|
|
52
|
+
ref_updated: float = 0.0
|
|
53
|
+
error: str = ""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass
|
|
57
|
+
class Attribution:
|
|
58
|
+
tasks: list[Task] = field(default_factory=list)
|
|
59
|
+
unattributed: dict[str, list[Request]] = field(default_factory=dict)
|
|
60
|
+
repos: dict[str, RepoInfo] = field(default_factory=dict)
|
|
61
|
+
# session_id -> repos it committed in, ran in, or edited files in
|
|
62
|
+
session_repos: dict[str, set[str]] = field(default_factory=dict)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def attribute(
|
|
66
|
+
parsed: ParseResult,
|
|
67
|
+
since: float,
|
|
68
|
+
now: float,
|
|
69
|
+
*,
|
|
70
|
+
refresh: bool = False,
|
|
71
|
+
find_root: Callable[[str], str | None] = repo_root,
|
|
72
|
+
) -> Attribution:
|
|
73
|
+
roots: dict[str, str | None] = {}
|
|
74
|
+
|
|
75
|
+
def root(path: str) -> str | None:
|
|
76
|
+
if path not in roots:
|
|
77
|
+
roots[path] = find_root(path)
|
|
78
|
+
return roots[path]
|
|
79
|
+
|
|
80
|
+
result = Attribution()
|
|
81
|
+
by_session: dict[str, list[Request]] = defaultdict(list)
|
|
82
|
+
for r in parsed.requests:
|
|
83
|
+
if r.ts >= since:
|
|
84
|
+
by_session[r.session_id].append(r)
|
|
85
|
+
|
|
86
|
+
calls_by_session: dict[str, list[tuple[str, CommitCall]]] = defaultdict(list)
|
|
87
|
+
for call in parsed.commit_calls:
|
|
88
|
+
end = call.end if call.end is not None else call.start
|
|
89
|
+
repo = root(call.cwd) if end >= since else None
|
|
90
|
+
if repo is not None:
|
|
91
|
+
calls_by_session[call.session_id].append((repo, call))
|
|
92
|
+
|
|
93
|
+
# Repos each session may have worked in: where it committed, where its
|
|
94
|
+
# requests ran, and where the files it edited live. A session's cwd is
|
|
95
|
+
# often not the repo it worked on (absolute paths, `cd x && ...`).
|
|
96
|
+
session_repos: dict[str, set[str]] = {}
|
|
97
|
+
for session, group in by_session.items():
|
|
98
|
+
repos = {repo for repo, _ in calls_by_session.get(session, [])}
|
|
99
|
+
repos |= {root(r.cwd) for r in group} - {None}
|
|
100
|
+
for cwd, path in parsed.files_touched.get(session, set()):
|
|
101
|
+
full = path if os.path.isabs(path) else os.path.join(cwd, path)
|
|
102
|
+
repos |= {root(os.path.dirname(os.path.normpath(full)))} - {None}
|
|
103
|
+
session_repos[session] = repos
|
|
104
|
+
result.session_repos = session_repos
|
|
105
|
+
|
|
106
|
+
commits_by_repo: dict[str, dict[str, Commit]] = {}
|
|
107
|
+
status_by_repo: dict[str, dict[str, str]] = {}
|
|
108
|
+
all_repos = sorted(set().union(*session_repos.values()) if session_repos else set())
|
|
109
|
+
# git subprocesses release the GIL, so repos load in parallel
|
|
110
|
+
with ThreadPoolExecutor(max_workers=min(8, len(all_repos) or 1)) as pool:
|
|
111
|
+
loaded = list(pool.map(lambda repo: _load_repo(repo, since - DAY, now, refresh), all_repos))
|
|
112
|
+
for repo, (commits, statuses, info) in zip(all_repos, loaded):
|
|
113
|
+
commits_by_repo[repo] = {c.sha: c for c in commits}
|
|
114
|
+
status_by_repo[repo] = statuses
|
|
115
|
+
result.repos[repo] = info
|
|
116
|
+
|
|
117
|
+
def task_for(repo: str, commit: Commit, attribution: str) -> Task:
|
|
118
|
+
key = (repo, commit.sha)
|
|
119
|
+
if key not in tasks:
|
|
120
|
+
tasks[key] = Task(
|
|
121
|
+
repo=repo,
|
|
122
|
+
sha=commit.sha,
|
|
123
|
+
ts=commit.ts,
|
|
124
|
+
subject=commit.subject,
|
|
125
|
+
attribution=attribution,
|
|
126
|
+
status=status_by_repo.get(repo, {}).get(commit.sha, GIT_FAILED),
|
|
127
|
+
)
|
|
128
|
+
return tasks[key]
|
|
129
|
+
|
|
130
|
+
# Tier 1: time segments between the session's exact commits, in any repo.
|
|
131
|
+
tasks: dict[tuple[str, str], Task] = {}
|
|
132
|
+
leftovers: dict[str, list[Request]] = {}
|
|
133
|
+
for session, group in by_session.items():
|
|
134
|
+
group.sort(key=lambda r: r.ts)
|
|
135
|
+
i = 0
|
|
136
|
+
for repo, commit in _anchors(calls_by_session.get(session, []), commits_by_repo):
|
|
137
|
+
task = task_for(repo, commit, "exact")
|
|
138
|
+
j = i
|
|
139
|
+
while j < len(group) and group[j].ts <= commit.ts + PAD:
|
|
140
|
+
j += 1
|
|
141
|
+
_add_bursts(task, group[i:j], 1.0)
|
|
142
|
+
i = j
|
|
143
|
+
if i < len(group):
|
|
144
|
+
leftovers[session] = group[i:]
|
|
145
|
+
|
|
146
|
+
# Tier 2 runs after every exact anchor is known, so a Claude-made commit
|
|
147
|
+
# is never also claimed as someone's hand-made commit.
|
|
148
|
+
exact = set(tasks)
|
|
149
|
+
emails: dict[str, str] = {}
|
|
150
|
+
for session, rest in leftovers.items():
|
|
151
|
+
matches: list[tuple[str, Commit]] = []
|
|
152
|
+
for repo in sorted(session_repos.get(session, set())):
|
|
153
|
+
if repo not in emails:
|
|
154
|
+
emails[repo] = user_email(repo)
|
|
155
|
+
touched = _repo_relative(parsed.files_touched.get(session, set()), repo)
|
|
156
|
+
matches += [(repo, c) for c in _fuzzy(rest, touched, commits_by_repo.get(repo, {}), exact, repo, emails[repo])]
|
|
157
|
+
if not matches:
|
|
158
|
+
reason = NO_COMMIT_YET if session_repos.get(session) else NOT_IN_REPO
|
|
159
|
+
result.unattributed.setdefault(reason, []).extend(rest)
|
|
160
|
+
continue
|
|
161
|
+
total_lines = sum(max(c.lines_changed, 1) for _, c in matches)
|
|
162
|
+
for repo, commit in matches:
|
|
163
|
+
task = task_for(repo, commit, "fuzzy" if commit.ts <= rest[-1].ts + PAD else "grace")
|
|
164
|
+
share = max(commit.lines_changed, 1) / total_lines
|
|
165
|
+
_add_bursts(task, rest, share)
|
|
166
|
+
|
|
167
|
+
result.tasks = sorted(tasks.values(), key=lambda t: t.ts)
|
|
168
|
+
return result
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _add_bursts(task: Task, segment: list[Request], weight: float) -> None:
|
|
172
|
+
"""The final work burst before a commit is its direct cost; earlier bursts are lead-up.
|
|
173
|
+
|
|
174
|
+
r r r ··· 2h+ break ··· r r ··· 5h break ··· r r r [commit]
|
|
175
|
+
└ lead-up ┘ └ lead-up ┘ └ direct ┘
|
|
176
|
+
"""
|
|
177
|
+
start = 0
|
|
178
|
+
for k in range(1, len(segment)):
|
|
179
|
+
if segment[k].ts - segment[k - 1].ts >= BREAK:
|
|
180
|
+
start = k
|
|
181
|
+
for k, r in enumerate(segment):
|
|
182
|
+
task.add(r, weight, lead_up=k < start)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _load_repo(repo: str, since: float, now: float, refresh: bool) -> tuple[list[Commit], dict[str, str], RepoInfo]:
|
|
186
|
+
info = RepoInfo()
|
|
187
|
+
commits: list[Commit] = []
|
|
188
|
+
statuses: dict[str, str] = {}
|
|
189
|
+
try:
|
|
190
|
+
commits = load_commits(repo, since)
|
|
191
|
+
checker = KeptChecker(repo, since, now, refresh=refresh, commits=commits)
|
|
192
|
+
info.ref, info.ref_updated = checker.ref, checker.ref_updated
|
|
193
|
+
statuses = {sha: s.value for sha, s in checker.classify(commits).items()}
|
|
194
|
+
except GitError as exc:
|
|
195
|
+
# Commits may still have loaded (e.g. no default branch): attribute
|
|
196
|
+
# them, and their status reads "git error" in the receipt.
|
|
197
|
+
info.error = str(exc)
|
|
198
|
+
return commits, statuses, info
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _repo_relative(touched: set[tuple[str, str]], repo: str) -> set[str]:
|
|
202
|
+
out = set()
|
|
203
|
+
for cwd, path in touched:
|
|
204
|
+
full = os.path.normpath(path if os.path.isabs(path) else os.path.join(cwd, path))
|
|
205
|
+
full = os.path.realpath(full) if os.path.exists(full) else full
|
|
206
|
+
rel = os.path.relpath(full, repo)
|
|
207
|
+
if not rel.startswith(".."):
|
|
208
|
+
out.add(rel)
|
|
209
|
+
return out
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _fuzzy(
|
|
213
|
+
rest: list[Request],
|
|
214
|
+
touched: set[str],
|
|
215
|
+
commits: dict[str, Commit],
|
|
216
|
+
exact: set[tuple[str, str]],
|
|
217
|
+
repo: str,
|
|
218
|
+
email: str,
|
|
219
|
+
) -> list[Commit]:
|
|
220
|
+
"""Your own hand-made commits in the session's window (+2h grace) that touch files it edited.
|
|
221
|
+
|
|
222
|
+
Authorship matters: after a pull, teammates' merged commits land inside the
|
|
223
|
+
window and touch the same files. Without a configured user.email we don't guess.
|
|
224
|
+
"""
|
|
225
|
+
if not touched or not email:
|
|
226
|
+
return []
|
|
227
|
+
start, end = rest[0].ts - PAD, rest[-1].ts + GRACE
|
|
228
|
+
return sorted(
|
|
229
|
+
(
|
|
230
|
+
c
|
|
231
|
+
for c in commits.values()
|
|
232
|
+
if start <= c.ts <= end
|
|
233
|
+
and c.author_email == email
|
|
234
|
+
and (repo, c.sha) not in exact
|
|
235
|
+
and touched & set(c.files)
|
|
236
|
+
),
|
|
237
|
+
key=lambda c: c.ts,
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _anchors(calls: list[tuple[str, CommitCall]], commits_by_repo: dict[str, dict[str, Commit]]) -> list[tuple[str, Commit]]:
|
|
242
|
+
"""(repo, commit) whose time falls inside one of the session's commit calls, oldest first."""
|
|
243
|
+
found: dict[tuple[str, str], tuple[str, Commit]] = {}
|
|
244
|
+
for repo, call in calls:
|
|
245
|
+
start = call.start - PAD
|
|
246
|
+
end = (call.end if call.end is not None else call.start) + PAD
|
|
247
|
+
for commit in commits_by_repo.get(repo, {}).values():
|
|
248
|
+
if start <= commit.ts <= end:
|
|
249
|
+
found[(repo, commit.sha)] = (repo, commit)
|
|
250
|
+
return sorted(found.values(), key=lambda rc: rc[1].ts)
|
|
251
|
+
|
|
252
|
+
|
aimpg/cli.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""`aimpg report`: print the fuel receipt for your local Claude Code usage."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
import time
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from aimpg.attribution import attribute
|
|
11
|
+
from aimpg.gitkept import DAY
|
|
12
|
+
from aimpg.logs import DEFAULT_ROOT, iter_log_files, parse_logs
|
|
13
|
+
from aimpg.receipt import render
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def main(argv: list[str] | None = None) -> int:
|
|
17
|
+
parser = argparse.ArgumentParser(prog="aimpg", description=__doc__)
|
|
18
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
19
|
+
report = sub.add_parser("report", help="print the energy receipt")
|
|
20
|
+
report.add_argument("--days", type=int, default=30, help="window size in days (default 30)")
|
|
21
|
+
report.add_argument("--logs", type=Path, default=DEFAULT_ROOT, help="Claude Code projects dir")
|
|
22
|
+
report.add_argument("--fetch", action="store_true", help="git fetch each repo first (uses the network)")
|
|
23
|
+
args = parser.parse_args(argv)
|
|
24
|
+
|
|
25
|
+
if args.days <= 0:
|
|
26
|
+
parser.error("--days must be positive")
|
|
27
|
+
if not args.logs.is_dir():
|
|
28
|
+
print(f"No Claude Code logs found at {args.logs}. Nothing to report.")
|
|
29
|
+
return 0
|
|
30
|
+
|
|
31
|
+
now = time.time()
|
|
32
|
+
since = now - args.days * DAY
|
|
33
|
+
parsed = parse_logs(iter_log_files(args.logs))
|
|
34
|
+
attribution = attribute(parsed, since, now, refresh=args.fetch)
|
|
35
|
+
sys.stdout.write(render(parsed, attribution, since, now))
|
|
36
|
+
return 0
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
if __name__ == "__main__":
|
|
40
|
+
raise SystemExit(main())
|
aimpg/energy.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""The one place energy is computed. Everything else calls this module.
|
|
2
|
+
|
|
3
|
+
Per request (joules, before overhead):
|
|
4
|
+
|
|
5
|
+
prefill = (fresh_in + cache_write) * E_pre # a cache write is a full prefill
|
|
6
|
+
cache = cache_read * E_kv # free if still in HBM, a reload if not
|
|
7
|
+
decode = output * (E_dec + E_ctx * ctx_len) # every output token re-reads the KV cache
|
|
8
|
+
|
|
9
|
+
E_pre = 2 * P * J_flop * prefill_inefficiency
|
|
10
|
+
E_dec = (P * weight_bits / batch) * J_bit + 2 * P * J_flop
|
|
11
|
+
E_ctx = kv_bits_per_token * J_bit
|
|
12
|
+
E_kv = kv_bits_per_token * J_bit * kv_reload_multiplier
|
|
13
|
+
|
|
14
|
+
Wh = overhead * joules / 3600
|
|
15
|
+
|
|
16
|
+
Every input is a [low, high] range from factors.json. The formula is
|
|
17
|
+
monotone increasing in each factor (taking batch_size reversed), so the
|
|
18
|
+
all-low and all-high corners bound the result.
|
|
19
|
+
|
|
20
|
+
Ranking (W) removes the model-class scale: W = joules / E_pre. Because the
|
|
21
|
+
ratios inside W are estimates, `compare` checks every corner of the factor
|
|
22
|
+
ranges and only names a winner when it wins at all of them.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import itertools
|
|
28
|
+
import json
|
|
29
|
+
from dataclasses import dataclass
|
|
30
|
+
from enum import Enum
|
|
31
|
+
from functools import lru_cache
|
|
32
|
+
from importlib import resources
|
|
33
|
+
from typing import Iterable
|
|
34
|
+
|
|
35
|
+
from aimpg.model import Request, Usage
|
|
36
|
+
|
|
37
|
+
_RANGE_KEYS = (
|
|
38
|
+
"batch_size",
|
|
39
|
+
"weight_bits",
|
|
40
|
+
"kv_bits_per_token",
|
|
41
|
+
"prefill_inefficiency",
|
|
42
|
+
"kv_reload_multiplier",
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class Factors:
|
|
48
|
+
params: float # active parameters (count, not billions)
|
|
49
|
+
batch_size: float
|
|
50
|
+
weight_bits: float
|
|
51
|
+
kv_bits_per_token: float
|
|
52
|
+
prefill_inefficiency: float
|
|
53
|
+
kv_reload_multiplier: float
|
|
54
|
+
overhead: float
|
|
55
|
+
j_flop: float
|
|
56
|
+
j_bit: float
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def e_pre(self) -> float:
|
|
60
|
+
return 2 * self.params * self.j_flop * self.prefill_inefficiency
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def e_dec(self) -> float:
|
|
64
|
+
return (self.params * self.weight_bits / self.batch_size) * self.j_bit + 2 * self.params * self.j_flop
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def e_ctx(self) -> float:
|
|
68
|
+
return self.kv_bits_per_token * self.j_bit
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def e_kv(self) -> float:
|
|
72
|
+
return self.kv_bits_per_token * self.j_bit * self.kv_reload_multiplier
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass(frozen=True)
|
|
76
|
+
class WhRange:
|
|
77
|
+
low: float
|
|
78
|
+
high: float
|
|
79
|
+
|
|
80
|
+
def __add__(self, other: "WhRange") -> "WhRange":
|
|
81
|
+
return WhRange(self.low + other.low, self.high + other.high)
|
|
82
|
+
|
|
83
|
+
def __mul__(self, k: float) -> "WhRange":
|
|
84
|
+
return WhRange(self.low * k, self.high * k)
|
|
85
|
+
|
|
86
|
+
@property
|
|
87
|
+
def mid(self) -> float:
|
|
88
|
+
return (self.low * self.high) ** 0.5 # geometric: the range spans orders of magnitude
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
ZERO = WhRange(0.0, 0.0)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@lru_cache(maxsize=1)
|
|
95
|
+
def load_factors() -> dict:
|
|
96
|
+
return json.loads(resources.files("aimpg").joinpath("factors.json").read_text())
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def model_class(model: str) -> tuple[str, bool]:
|
|
100
|
+
"""(class name, assumed) — assumed is True when the name matched no class."""
|
|
101
|
+
data = load_factors()
|
|
102
|
+
name = model.lower()
|
|
103
|
+
for cls, spec in data["classes"].items():
|
|
104
|
+
if any(token in name for token in spec["matches"]):
|
|
105
|
+
return cls, False
|
|
106
|
+
return data["default_class"], True
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def corners(cls: str) -> list[Factors]:
|
|
110
|
+
"""Every combination of low/high for the uncertain factors, for one class."""
|
|
111
|
+
data = load_factors()
|
|
112
|
+
ranges = data["ranges"]
|
|
113
|
+
consts = data["constants"]
|
|
114
|
+
params = data["classes"][cls]["active_params_billion"]
|
|
115
|
+
out = []
|
|
116
|
+
axes = [("params", params)] + [(k, ranges[k]) for k in _RANGE_KEYS] + [("overhead", ranges["overhead"])]
|
|
117
|
+
for picks in itertools.product(("low", "high"), repeat=len(axes)):
|
|
118
|
+
values = {name: spec[pick] for (name, spec), pick in zip(axes, picks)}
|
|
119
|
+
values["params"] *= 1e9
|
|
120
|
+
out.append(Factors(j_flop=consts["joules_per_flop"]["value"], j_bit=consts["joules_per_hbm_bit"]["value"], **values))
|
|
121
|
+
return out
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _end(cls: str, which: str) -> Factors:
|
|
125
|
+
data = load_factors()
|
|
126
|
+
ranges = data["ranges"]
|
|
127
|
+
consts = data["constants"]
|
|
128
|
+
return Factors(
|
|
129
|
+
params=data["classes"][cls]["active_params_billion"][which] * 1e9,
|
|
130
|
+
overhead=ranges["overhead"][which],
|
|
131
|
+
j_flop=consts["joules_per_flop"]["value"],
|
|
132
|
+
j_bit=consts["joules_per_hbm_bit"]["value"],
|
|
133
|
+
**{k: ranges[k][which] for k in _RANGE_KEYS},
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def joules(usage: Usage, f: Factors) -> float:
|
|
138
|
+
"""GPU-side joules for one request, before datacenter overhead."""
|
|
139
|
+
prefill = (usage.fresh_in + usage.cache_write) * f.e_pre
|
|
140
|
+
cache = usage.cache_read * f.e_kv
|
|
141
|
+
decode = usage.output * (f.e_dec + f.e_ctx * usage.ctx_len)
|
|
142
|
+
return prefill + cache + decode
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def wh(usage: Usage, f: Factors) -> float:
|
|
146
|
+
return f.overhead * joules(usage, f) / 3600.0
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def request_wh(request: Request) -> WhRange:
|
|
150
|
+
cls, _ = model_class(request.model)
|
|
151
|
+
return WhRange(wh(request.usage, _end(cls, "low")), wh(request.usage, _end(cls, "high")))
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def total_wh(requests: Iterable[Request]) -> WhRange:
|
|
155
|
+
total = ZERO
|
|
156
|
+
for r in requests:
|
|
157
|
+
total = total + request_wh(r)
|
|
158
|
+
return total
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def weighted_wh(requests: Iterable[Request], weights: Iterable[float]) -> WhRange:
|
|
162
|
+
total = ZERO
|
|
163
|
+
for r, w in zip(requests, weights):
|
|
164
|
+
total = total + request_wh(r) * w
|
|
165
|
+
return total
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def weighted_tokens(usages: Iterable[Usage], f: Factors) -> float:
|
|
169
|
+
"""W: joules with the model-class scale (E_pre) divided out."""
|
|
170
|
+
return sum(joules(u, f) for u in usages) / f.e_pre
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
class Verdict(str, Enum):
|
|
174
|
+
A_WINS = "A uses less"
|
|
175
|
+
B_WINS = "B uses less"
|
|
176
|
+
TOO_CLOSE = "too close to call with current energy data"
|
|
177
|
+
NOT_COMPARABLE = "not directly comparable (different model classes)"
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def compare(a: list[Request], b: list[Request]) -> Verdict:
|
|
181
|
+
"""Same-model-class comparison that holds at every corner of the factor ranges."""
|
|
182
|
+
classes = {model_class(r.model)[0] for r in a + b}
|
|
183
|
+
if len(classes) != 1:
|
|
184
|
+
return Verdict.NOT_COMPARABLE
|
|
185
|
+
(cls,) = classes
|
|
186
|
+
ua, ub = [r.usage for r in a], [r.usage for r in b]
|
|
187
|
+
signs = {
|
|
188
|
+
(weighted_tokens(ua, f) > weighted_tokens(ub, f)) - (weighted_tokens(ua, f) < weighted_tokens(ub, f))
|
|
189
|
+
for f in corners(cls)
|
|
190
|
+
}
|
|
191
|
+
if signs == {-1}:
|
|
192
|
+
return Verdict.A_WINS
|
|
193
|
+
if signs == {1}:
|
|
194
|
+
return Verdict.B_WINS
|
|
195
|
+
return Verdict.TOO_CLOSE
|
aimpg/factors.json
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": "2026-10-01",
|
|
3
|
+
"constants": {
|
|
4
|
+
"joules_per_flop": {
|
|
5
|
+
"value": 0.52e-12,
|
|
6
|
+
"citation": "From Tokens to Watt-hours (arXiv 2607.26571): tensor-core coefficient alpha_TC = 0.52 pJ/FLOP, H100 BF16"
|
|
7
|
+
},
|
|
8
|
+
"joules_per_hbm_bit": {
|
|
9
|
+
"value": 11.68e-12,
|
|
10
|
+
"citation": "From Tokens to Watt-hours (arXiv 2607.26571): HBM energy e_HBM = 11.68 pJ/bit, H100"
|
|
11
|
+
}
|
|
12
|
+
},
|
|
13
|
+
"ranges": {
|
|
14
|
+
"overhead": {
|
|
15
|
+
"low": 1.25,
|
|
16
|
+
"high": 1.5,
|
|
17
|
+
"citation": "EcoLogits LLM inference methodology: non-GPU server power 1.2 kW per 8-GPU server, provider PUE 1.09-1.20"
|
|
18
|
+
},
|
|
19
|
+
"batch_size": {
|
|
20
|
+
"low": 128,
|
|
21
|
+
"high": 32,
|
|
22
|
+
"citation": "Assumption. EcoLogits fixes B=64; we bracket it. Larger batches amortize weight reads, so 128 is the low-energy end"
|
|
23
|
+
},
|
|
24
|
+
"weight_bits": {
|
|
25
|
+
"low": 8,
|
|
26
|
+
"high": 16,
|
|
27
|
+
"citation": "Assumption: FP8 to BF16 serving precision"
|
|
28
|
+
},
|
|
29
|
+
"kv_bits_per_token": {
|
|
30
|
+
"low": 0.5e6,
|
|
31
|
+
"high": 3.0e6,
|
|
32
|
+
"citation": "Assumption: 2 x layers x kv_heads x head_dim x bytes for GQA models (e.g. 80x8x128x2x2 bytes = 2.6 Mbit BF16); FP8/compressed caches are lower"
|
|
33
|
+
},
|
|
34
|
+
"prefill_inefficiency": {
|
|
35
|
+
"low": 1.0,
|
|
36
|
+
"high": 2.0,
|
|
37
|
+
"citation": "Assumption: static power and sub-peak utilization during prefill on top of pure FLOP energy"
|
|
38
|
+
},
|
|
39
|
+
"kv_reload_multiplier": {
|
|
40
|
+
"low": 0.0,
|
|
41
|
+
"high": 4.0,
|
|
42
|
+
"citation": "Assumption: a cache read is free if the prefix is still in HBM (low) or costs a reload from host memory at about 4x HBM read energy (high)"
|
|
43
|
+
}
|
|
44
|
+
},
|
|
45
|
+
"classes": {
|
|
46
|
+
"large": {
|
|
47
|
+
"active_params_billion": {"low": 100, "high": 400},
|
|
48
|
+
"matches": ["opus", "fable"],
|
|
49
|
+
"citation": "Assumption: parameter counts are not disclosed for frontier closed models"
|
|
50
|
+
},
|
|
51
|
+
"mid": {
|
|
52
|
+
"active_params_billion": {"low": 30, "high": 150},
|
|
53
|
+
"matches": ["sonnet"],
|
|
54
|
+
"citation": "Assumption: parameter counts are not disclosed for frontier closed models"
|
|
55
|
+
},
|
|
56
|
+
"small": {
|
|
57
|
+
"active_params_billion": {"low": 8, "high": 40},
|
|
58
|
+
"matches": ["haiku"],
|
|
59
|
+
"citation": "Assumption: parameter counts are not disclosed for frontier closed models"
|
|
60
|
+
}
|
|
61
|
+
},
|
|
62
|
+
"default_class": "mid"
|
|
63
|
+
}
|