aimpg 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aimpg-0.1.0/.github/workflows/publish.yml +17 -0
- aimpg-0.1.0/.github/workflows/test.yml +17 -0
- aimpg-0.1.0/.gitignore +6 -0
- aimpg-0.1.0/LICENSE +21 -0
- aimpg-0.1.0/PKG-INFO +93 -0
- aimpg-0.1.0/README.md +73 -0
- aimpg-0.1.0/TODOS.md +27 -0
- aimpg-0.1.0/aimpg/__init__.py +1 -0
- aimpg-0.1.0/aimpg/attribution.py +252 -0
- aimpg-0.1.0/aimpg/cli.py +40 -0
- aimpg-0.1.0/aimpg/energy.py +195 -0
- aimpg-0.1.0/aimpg/factors.json +63 -0
- aimpg-0.1.0/aimpg/gitkept.py +255 -0
- aimpg-0.1.0/aimpg/logs.py +318 -0
- aimpg-0.1.0/aimpg/model.py +97 -0
- aimpg-0.1.0/aimpg/receipt.py +149 -0
- aimpg-0.1.0/docs/designs/aimpg-design.md +275 -0
- aimpg-0.1.0/evals/attribution_eval.py +100 -0
- aimpg-0.1.0/evals/label.py +165 -0
- aimpg-0.1.0/pyproject.toml +41 -0
- aimpg-0.1.0/tests/conftest.py +83 -0
- aimpg-0.1.0/tests/gitrepo.py +47 -0
- aimpg-0.1.0/tests/test_attribution.py +294 -0
- aimpg-0.1.0/tests/test_energy.py +104 -0
- aimpg-0.1.0/tests/test_eval_scoring.py +45 -0
- aimpg-0.1.0/tests/test_gitkept.py +120 -0
- aimpg-0.1.0/tests/test_logs.py +175 -0
- aimpg-0.1.0/tests/test_perf.py +81 -0
- aimpg-0.1.0/uv.lock +156 -0
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
name: publish
|
|
2
|
+
on:
|
|
3
|
+
push:
|
|
4
|
+
tags: ["v*"]
|
|
5
|
+
jobs:
|
|
6
|
+
publish:
|
|
7
|
+
runs-on: ubuntu-latest
|
|
8
|
+
environment: pypi
|
|
9
|
+
permissions:
|
|
10
|
+
id-token: write # PyPI trusted publishing, no API token stored
|
|
11
|
+
steps:
|
|
12
|
+
- uses: actions/checkout@v4
|
|
13
|
+
- uses: astral-sh/setup-uv@v6
|
|
14
|
+
- run: git config --global user.email ci@example.com && git config --global user.name CI
|
|
15
|
+
- run: uv run --group dev pytest -q
|
|
16
|
+
- run: uv build
|
|
17
|
+
- run: uv publish --trusted-publishing always
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
name: test
|
|
2
|
+
on:
|
|
3
|
+
push:
|
|
4
|
+
branches: [main]
|
|
5
|
+
pull_request:
|
|
6
|
+
jobs:
|
|
7
|
+
test:
|
|
8
|
+
strategy:
|
|
9
|
+
matrix:
|
|
10
|
+
os: [ubuntu-latest, macos-latest]
|
|
11
|
+
python: ["3.10", "3.13"]
|
|
12
|
+
runs-on: ${{ matrix.os }}
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v4
|
|
15
|
+
- uses: astral-sh/setup-uv@v6
|
|
16
|
+
- run: git config --global user.email ci@example.com && git config --global user.name CI
|
|
17
|
+
- run: uv run --python ${{ matrix.python }} --group dev pytest -q
|
aimpg-0.1.0/.gitignore
ADDED
aimpg-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Kumar Ganduri
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
aimpg-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: aimpg
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Real-world energy per solved task for AI coding agents
|
|
5
|
+
Project-URL: Homepage, https://github.com/kumarganduri/aimpg
|
|
6
|
+
Project-URL: Issues, https://github.com/kumarganduri/aimpg/issues
|
|
7
|
+
Author: Kumar Ganduri
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: ai,carbon,claude-code,energy,llm,sustainability
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: MacOS
|
|
15
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Topic :: Software Development
|
|
18
|
+
Requires-Python: >=3.10
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
|
|
21
|
+
# aimpg
|
|
22
|
+
|
|
23
|
+
**Miles per gallon for AI coding.** Find out how much energy your AI coding agent used, and which of your commits it went into.
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
uvx aimpg report
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Example output:
|
|
30
|
+
|
|
31
|
+
```
|
|
32
|
+
AI energy in window .......................... 6.99 kWh – 57.39 kWh (4,744 requests)
|
|
33
|
+
matched to commits ......................... 5.87 kWh – 47.96 kWh (223 commits: 223 exact, 0 fuzzy, 0 grace)
|
|
34
|
+
exact-match share of in-repo energy: 84%, any match: 84%
|
|
35
|
+
no commit from this session yet ............ 1.11 kWh – 9.38 kWh (609 requests)
|
|
36
|
+
|
|
37
|
+
Median direct energy per kept commit: 27.2 Wh – 212.9 Wh
|
|
38
|
+
|
|
39
|
+
Most energy-hungry commits:
|
|
40
|
+
213.2 Wh – 1.71 kWh my-app 2ef4297 fix: date parsing for day-first locales
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## What it does
|
|
44
|
+
|
|
45
|
+
`aimpg` reads the Claude Code session logs already on your machine (`~/.claude/projects`). It works out which AI requests produced which of your git commits, and prints an energy receipt:
|
|
46
|
+
|
|
47
|
+
- **Energy per kept commit**, as a low–high range in Wh
|
|
48
|
+
- **Discarded work**: energy that went into commits that never reached your main branch
|
|
49
|
+
- **The most and least energy-hungry commits**
|
|
50
|
+
- **Unmatched energy**: work that produced no commit (yet), shown openly rather than hidden
|
|
51
|
+
|
|
52
|
+
## Privacy
|
|
53
|
+
|
|
54
|
+
Everything runs locally. `aimpg report` makes no network calls (unless you pass `--fetch`, which runs `git fetch`). Nothing is uploaded, and your code and prompts never leave your machine.
|
|
55
|
+
|
|
56
|
+
## How it matches requests to commits
|
|
57
|
+
|
|
58
|
+
1. **Exact:** when the agent runs `git commit`, the commit lands while that tool call is running. We match commits to those call windows, including slow commits where pre-commit hooks run for minutes.
|
|
59
|
+
2. **Time segments:** within a session, the requests made since the previous commit belong to the next one, in whatever repo it lands. A 2h+ break starts a new work burst. Only the final burst counts as the commit's *direct* energy, and earlier bursts are shown as *lead-up*.
|
|
60
|
+
3. **Fuzzy (fallback):** commits you make by hand are matched only if you authored them (your `user.email`), within 2 hours of the session, and only if they touch files the session edited. Teammates' commits are never claimed.
|
|
61
|
+
|
|
62
|
+
On the author's own history, a hand-labeled check of 20 commits matched 20/20 to the right session (`evals/`).
|
|
63
|
+
|
|
64
|
+
## How energy is estimated
|
|
65
|
+
|
|
66
|
+
Model sizes for Claude aren't public, so every number is a **range**, never a single figure. The formula is physical, not price-based:
|
|
67
|
+
|
|
68
|
+
- **Prefill:** compute energy for every fresh or cache-written input token.
|
|
69
|
+
- **Decode:** each output token re-reads the model weights and the whole KV cache, so long contexts make every output token more expensive.
|
|
70
|
+
- **Cache reads:** free when the cache is still in GPU memory, a reload when it isn't.
|
|
71
|
+
- **Overhead:** server and datacenter overhead (PUE) on top.
|
|
72
|
+
|
|
73
|
+
Per-operation energy comes from [From Tokens to Watt-hours](https://arxiv.org/html/2607.26571v1). Server overhead and PUE come from [EcoLogits](https://ecologits.ai/latest/methodology/llm_inference/). Model-size classes are labeled assumptions. Every coefficient and its source is in [`aimpg/factors.json`](aimpg/factors.json).
|
|
74
|
+
|
|
75
|
+
## Options
|
|
76
|
+
|
|
77
|
+
```
|
|
78
|
+
aimpg report [--days 30] [--logs ~/.claude/projects] [--fetch]
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Limits (honest list)
|
|
82
|
+
|
|
83
|
+
- Claude Code logs only, for now.
|
|
84
|
+
- Work before a 2h+ break is reported as a commit's *lead-up*, separate from its *direct* energy. A multi-day feature that genuinely needed that earlier work will look cheaper in the direct number, so check the lead-up too.
|
|
85
|
+
- Commits less than 7 days old show as `pending` until we can tell whether they were kept.
|
|
86
|
+
|
|
87
|
+
## Roadmap
|
|
88
|
+
|
|
89
|
+
Next is `aimpg replay`: rerun your own past commits through different setups (for example Claude Code with and without a token saver) to prove which one saves energy without breaking your tests. See [docs/designs/aimpg-design.md](docs/designs/aimpg-design.md).
|
|
90
|
+
|
|
91
|
+
## License
|
|
92
|
+
|
|
93
|
+
MIT
|
aimpg-0.1.0/README.md
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
# aimpg
|
|
2
|
+
|
|
3
|
+
**Miles per gallon for AI coding.** Find out how much energy your AI coding agent used, and which of your commits it went into.
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
uvx aimpg report
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
Example output:
|
|
10
|
+
|
|
11
|
+
```
|
|
12
|
+
AI energy in window .......................... 6.99 kWh – 57.39 kWh (4,744 requests)
|
|
13
|
+
matched to commits ......................... 5.87 kWh – 47.96 kWh (223 commits: 223 exact, 0 fuzzy, 0 grace)
|
|
14
|
+
exact-match share of in-repo energy: 84%, any match: 84%
|
|
15
|
+
no commit from this session yet ............ 1.11 kWh – 9.38 kWh (609 requests)
|
|
16
|
+
|
|
17
|
+
Median direct energy per kept commit: 27.2 Wh – 212.9 Wh
|
|
18
|
+
|
|
19
|
+
Most energy-hungry commits:
|
|
20
|
+
213.2 Wh – 1.71 kWh my-app 2ef4297 fix: date parsing for day-first locales
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## What it does
|
|
24
|
+
|
|
25
|
+
`aimpg` reads the Claude Code session logs already on your machine (`~/.claude/projects`). It works out which AI requests produced which of your git commits, and prints an energy receipt:
|
|
26
|
+
|
|
27
|
+
- **Energy per kept commit**, as a low–high range in Wh
|
|
28
|
+
- **Discarded work**: energy that went into commits that never reached your main branch
|
|
29
|
+
- **The most and least energy-hungry commits**
|
|
30
|
+
- **Unmatched energy**: work that produced no commit (yet), shown openly rather than hidden
|
|
31
|
+
|
|
32
|
+
## Privacy
|
|
33
|
+
|
|
34
|
+
Everything runs locally. `aimpg report` makes no network calls (unless you pass `--fetch`, which runs `git fetch`). Nothing is uploaded, and your code and prompts never leave your machine.
|
|
35
|
+
|
|
36
|
+
## How it matches requests to commits
|
|
37
|
+
|
|
38
|
+
1. **Exact:** when the agent runs `git commit`, the commit lands while that tool call is running. We match commits to those call windows, including slow commits where pre-commit hooks run for minutes.
|
|
39
|
+
2. **Time segments:** within a session, the requests made since the previous commit belong to the next one, in whatever repo it lands. A 2h+ break starts a new work burst. Only the final burst counts as the commit's *direct* energy, and earlier bursts are shown as *lead-up*.
|
|
40
|
+
3. **Fuzzy (fallback):** commits you make by hand are matched only if you authored them (your `user.email`), within 2 hours of the session, and only if they touch files the session edited. Teammates' commits are never claimed.
|
|
41
|
+
|
|
42
|
+
On the author's own history, a hand-labeled check of 20 commits matched 20/20 to the right session (`evals/`).
|
|
43
|
+
|
|
44
|
+
## How energy is estimated
|
|
45
|
+
|
|
46
|
+
Model sizes for Claude aren't public, so every number is a **range**, never a single figure. The formula is physical, not price-based:
|
|
47
|
+
|
|
48
|
+
- **Prefill:** compute energy for every fresh or cache-written input token.
|
|
49
|
+
- **Decode:** each output token re-reads the model weights and the whole KV cache, so long contexts make every output token more expensive.
|
|
50
|
+
- **Cache reads:** free when the cache is still in GPU memory, a reload when it isn't.
|
|
51
|
+
- **Overhead:** server and datacenter overhead (PUE) on top.
|
|
52
|
+
|
|
53
|
+
Per-operation energy comes from [From Tokens to Watt-hours](https://arxiv.org/html/2607.26571v1). Server overhead and PUE come from [EcoLogits](https://ecologits.ai/latest/methodology/llm_inference/). Model-size classes are labeled assumptions. Every coefficient and its source is in [`aimpg/factors.json`](aimpg/factors.json).
|
|
54
|
+
|
|
55
|
+
## Options
|
|
56
|
+
|
|
57
|
+
```
|
|
58
|
+
aimpg report [--days 30] [--logs ~/.claude/projects] [--fetch]
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## Limits (honest list)
|
|
62
|
+
|
|
63
|
+
- Claude Code logs only, for now.
|
|
64
|
+
- Work before a 2h+ break is reported as a commit's *lead-up*, separate from its *direct* energy. A multi-day feature that genuinely needed that earlier work will look cheaper in the direct number, so check the lead-up too.
|
|
65
|
+
- Commits less than 7 days old show as `pending` until we can tell whether they were kept.
|
|
66
|
+
|
|
67
|
+
## Roadmap
|
|
68
|
+
|
|
69
|
+
Next is `aimpg replay`: rerun your own past commits through different setups (for example Claude Code with and without a token saver) to prove which one saves energy without breaking your tests. See [docs/designs/aimpg-design.md](docs/designs/aimpg-design.md).
|
|
70
|
+
|
|
71
|
+
## License
|
|
72
|
+
|
|
73
|
+
MIT
|
aimpg-0.1.0/TODOS.md
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# TODOS
|
|
2
|
+
|
|
3
|
+
## Phase 2 (blocking the Phase 2 eng review)
|
|
4
|
+
|
|
5
|
+
### Replay test leakage + model memorization
|
|
6
|
+
- **What:** Run replays SWE-bench style. Copy the commit's test files into the clone only after the agent finishes, then run them as the judge. Prefer commits newer than the model's training cutoff, or private repos.
|
|
7
|
+
- **Why:** At the parent commit, the commit's own tests don't exist yet. Adding them first leaks the answer; leaving them out means nothing judges pass/fail. Public repos (for example Legwork) may be in the model's training data, which inflates pass rates.
|
|
8
|
+
- **Pros:** Replay pass rates become trustworthy, and the whole leaderboard rests on them.
|
|
9
|
+
- **Cons:** Commits without test changes need another judge (a build check or a judge model), and the cutoff filter shrinks the candidate pool.
|
|
10
|
+
- **Context:** Raised by the outside voice in the 2026-10-01 eng review (finding 2). The design doc's Phase 2 step 2 says "run the repo's tests (or the tests touched by the commit)" without specifying when they're applied.
|
|
11
|
+
- **Depends on:** Phase 1 shipped; Phase 2 eng review.
|
|
12
|
+
|
|
13
|
+
### Statistical stability gate
|
|
14
|
+
- **What:** Replace "same order across 3 repeats" with a paired per-commit difference (setup A minus setup B) and a bootstrap 95% CI that must exclude 0. Re-cost the replay budget for 10+ commits.
|
|
15
|
+
- **Why:** With 2 setups, "same order 3 times" happens by chance 25% of the time, so the Phase 3 gate would pass on noise.
|
|
16
|
+
- **Pros:** Public rankings only appear when the difference is real.
|
|
17
|
+
- **Cons:** More replays means more token spend; the budget is likely above the current $60-150 estimate.
|
|
18
|
+
- **Context:** Outside voice finding 5, 2026-10-01 eng review. It affects Success Criteria (Phase 2) and the Phase 3 gate.
|
|
19
|
+
- **Depends on:** Phase 2 harness.
|
|
20
|
+
|
|
21
|
+
### Network-limit spike for the replay sandbox
|
|
22
|
+
- **What:** A time-boxed spike to make the replay container reach only the model API. Docker can't restrict outbound traffic by domain on its own, so it needs an allowlisting CONNECT proxy on an internal Docker network. Check what Legwork's sandbox actually does first.
|
|
23
|
+
- **Why:** The approved sandbox contract (eng review D7) promises "egress limited to the model API" but had no build plan or estimate.
|
|
24
|
+
- **Pros:** Makes the "no GitHub cheating" promise real.
|
|
25
|
+
- **Cons:** TLS through a proxy and the CLI's proxy settings can be fiddly per agent CLI.
|
|
26
|
+
- **Context:** Outside voice finding 7. Estimate: human ~1-2 days / CC ~2 hrs. It should be the first task of Phase 2.
|
|
27
|
+
- **Depends on:** Nothing. It can be spiked any time.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""aimpg: real-world energy per solved task for AI coding agents."""
|
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
"""Tie AI requests to the commits they produced.
|
|
2
|
+
|
|
3
|
+
Tier 1 (exact): a Bash tool call ran `git commit`, and a commit in that
|
|
4
|
+
repo has a committer time inside the call's [start, end] interval (±2s
|
|
5
|
+
for whole-second git timestamps). Measured 8/8 on real commit history,
|
|
6
|
+
including a commit whose hooks made it land 147s after the call started.
|
|
7
|
+
|
|
8
|
+
Energy split (time segments): within one session, the requests made after
|
|
9
|
+
commit k-1 and up to commit k belong to commit k, whatever repo they ran
|
|
10
|
+
in. A session's cwd is often not the repo it works on (absolute paths,
|
|
11
|
+
`cd x && ...`), so grouping by folder would strand requests.
|
|
12
|
+
|
|
13
|
+
session s: r r r [c1 in A] r r r r [c2 in B] r r
|
|
14
|
+
└── c1 ───┘ └──── c2 ─────┘ └ no commit yet
|
|
15
|
+
|
|
16
|
+
Long sessions: a 2h+ pause starts a new work burst. Only the final burst
|
|
17
|
+
before a commit is its direct energy; earlier bursts since the previous
|
|
18
|
+
commit are reported as its "lead-up", never dropped.
|
|
19
|
+
|
|
20
|
+
Tier 2 (fuzzy): a session's leftover requests (after its last exact
|
|
21
|
+
commit) go to commits in any repo it worked in, made between its first leftover
|
|
22
|
+
request and 2h after its last, authored by you (repo user.email) that touch a file the session edited
|
|
23
|
+
(Edit/Write tools or Bash: sed -i, redirects, mv/cp/rm). Several matches
|
|
24
|
+
split the requests by lines changed. Commits after the session's last
|
|
25
|
+
request are labeled "grace". Built because the first real run showed exact
|
|
26
|
+
matches covered only ~48% of in-repo energy (design gate: 70%).
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import os
|
|
32
|
+
from collections import defaultdict
|
|
33
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
34
|
+
from dataclasses import dataclass, field
|
|
35
|
+
from typing import Callable
|
|
36
|
+
|
|
37
|
+
from aimpg.gitkept import DAY, Commit, GitError, KeptChecker, load_commits, repo_root, user_email
|
|
38
|
+
from aimpg.model import CommitCall, ParseResult, Request, Task
|
|
39
|
+
|
|
40
|
+
PAD = 2.0 # seconds; git timestamps are whole seconds
|
|
41
|
+
GRACE = 2 * 3600.0
|
|
42
|
+
BREAK = 2 * 3600.0 # a pause this long between requests starts a new work burst # hand commits often land a while after the session goes quiet
|
|
43
|
+
|
|
44
|
+
NOT_IN_REPO = "not in a git repo (or repo moved/deleted)"
|
|
45
|
+
NO_COMMIT_YET = "no commit from this session yet"
|
|
46
|
+
GIT_FAILED = "git error"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass
|
|
50
|
+
class RepoInfo:
|
|
51
|
+
ref: str = ""
|
|
52
|
+
ref_updated: float = 0.0
|
|
53
|
+
error: str = ""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass
|
|
57
|
+
class Attribution:
|
|
58
|
+
tasks: list[Task] = field(default_factory=list)
|
|
59
|
+
unattributed: dict[str, list[Request]] = field(default_factory=dict)
|
|
60
|
+
repos: dict[str, RepoInfo] = field(default_factory=dict)
|
|
61
|
+
# session_id -> repos it committed in, ran in, or edited files in
|
|
62
|
+
session_repos: dict[str, set[str]] = field(default_factory=dict)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def attribute(
|
|
66
|
+
parsed: ParseResult,
|
|
67
|
+
since: float,
|
|
68
|
+
now: float,
|
|
69
|
+
*,
|
|
70
|
+
refresh: bool = False,
|
|
71
|
+
find_root: Callable[[str], str | None] = repo_root,
|
|
72
|
+
) -> Attribution:
|
|
73
|
+
roots: dict[str, str | None] = {}
|
|
74
|
+
|
|
75
|
+
def root(path: str) -> str | None:
|
|
76
|
+
if path not in roots:
|
|
77
|
+
roots[path] = find_root(path)
|
|
78
|
+
return roots[path]
|
|
79
|
+
|
|
80
|
+
result = Attribution()
|
|
81
|
+
by_session: dict[str, list[Request]] = defaultdict(list)
|
|
82
|
+
for r in parsed.requests:
|
|
83
|
+
if r.ts >= since:
|
|
84
|
+
by_session[r.session_id].append(r)
|
|
85
|
+
|
|
86
|
+
calls_by_session: dict[str, list[tuple[str, CommitCall]]] = defaultdict(list)
|
|
87
|
+
for call in parsed.commit_calls:
|
|
88
|
+
end = call.end if call.end is not None else call.start
|
|
89
|
+
repo = root(call.cwd) if end >= since else None
|
|
90
|
+
if repo is not None:
|
|
91
|
+
calls_by_session[call.session_id].append((repo, call))
|
|
92
|
+
|
|
93
|
+
# Repos each session may have worked in: where it committed, where its
|
|
94
|
+
# requests ran, and where the files it edited live. A session's cwd is
|
|
95
|
+
# often not the repo it worked on (absolute paths, `cd x && ...`).
|
|
96
|
+
session_repos: dict[str, set[str]] = {}
|
|
97
|
+
for session, group in by_session.items():
|
|
98
|
+
repos = {repo for repo, _ in calls_by_session.get(session, [])}
|
|
99
|
+
repos |= {root(r.cwd) for r in group} - {None}
|
|
100
|
+
for cwd, path in parsed.files_touched.get(session, set()):
|
|
101
|
+
full = path if os.path.isabs(path) else os.path.join(cwd, path)
|
|
102
|
+
repos |= {root(os.path.dirname(os.path.normpath(full)))} - {None}
|
|
103
|
+
session_repos[session] = repos
|
|
104
|
+
result.session_repos = session_repos
|
|
105
|
+
|
|
106
|
+
commits_by_repo: dict[str, dict[str, Commit]] = {}
|
|
107
|
+
status_by_repo: dict[str, dict[str, str]] = {}
|
|
108
|
+
all_repos = sorted(set().union(*session_repos.values()) if session_repos else set())
|
|
109
|
+
# git subprocesses release the GIL, so repos load in parallel
|
|
110
|
+
with ThreadPoolExecutor(max_workers=min(8, len(all_repos) or 1)) as pool:
|
|
111
|
+
loaded = list(pool.map(lambda repo: _load_repo(repo, since - DAY, now, refresh), all_repos))
|
|
112
|
+
for repo, (commits, statuses, info) in zip(all_repos, loaded):
|
|
113
|
+
commits_by_repo[repo] = {c.sha: c for c in commits}
|
|
114
|
+
status_by_repo[repo] = statuses
|
|
115
|
+
result.repos[repo] = info
|
|
116
|
+
|
|
117
|
+
def task_for(repo: str, commit: Commit, attribution: str) -> Task:
|
|
118
|
+
key = (repo, commit.sha)
|
|
119
|
+
if key not in tasks:
|
|
120
|
+
tasks[key] = Task(
|
|
121
|
+
repo=repo,
|
|
122
|
+
sha=commit.sha,
|
|
123
|
+
ts=commit.ts,
|
|
124
|
+
subject=commit.subject,
|
|
125
|
+
attribution=attribution,
|
|
126
|
+
status=status_by_repo.get(repo, {}).get(commit.sha, GIT_FAILED),
|
|
127
|
+
)
|
|
128
|
+
return tasks[key]
|
|
129
|
+
|
|
130
|
+
# Tier 1: time segments between the session's exact commits, in any repo.
|
|
131
|
+
tasks: dict[tuple[str, str], Task] = {}
|
|
132
|
+
leftovers: dict[str, list[Request]] = {}
|
|
133
|
+
for session, group in by_session.items():
|
|
134
|
+
group.sort(key=lambda r: r.ts)
|
|
135
|
+
i = 0
|
|
136
|
+
for repo, commit in _anchors(calls_by_session.get(session, []), commits_by_repo):
|
|
137
|
+
task = task_for(repo, commit, "exact")
|
|
138
|
+
j = i
|
|
139
|
+
while j < len(group) and group[j].ts <= commit.ts + PAD:
|
|
140
|
+
j += 1
|
|
141
|
+
_add_bursts(task, group[i:j], 1.0)
|
|
142
|
+
i = j
|
|
143
|
+
if i < len(group):
|
|
144
|
+
leftovers[session] = group[i:]
|
|
145
|
+
|
|
146
|
+
# Tier 2 runs after every exact anchor is known, so a Claude-made commit
|
|
147
|
+
# is never also claimed as someone's hand-made commit.
|
|
148
|
+
exact = set(tasks)
|
|
149
|
+
emails: dict[str, str] = {}
|
|
150
|
+
for session, rest in leftovers.items():
|
|
151
|
+
matches: list[tuple[str, Commit]] = []
|
|
152
|
+
for repo in sorted(session_repos.get(session, set())):
|
|
153
|
+
if repo not in emails:
|
|
154
|
+
emails[repo] = user_email(repo)
|
|
155
|
+
touched = _repo_relative(parsed.files_touched.get(session, set()), repo)
|
|
156
|
+
matches += [(repo, c) for c in _fuzzy(rest, touched, commits_by_repo.get(repo, {}), exact, repo, emails[repo])]
|
|
157
|
+
if not matches:
|
|
158
|
+
reason = NO_COMMIT_YET if session_repos.get(session) else NOT_IN_REPO
|
|
159
|
+
result.unattributed.setdefault(reason, []).extend(rest)
|
|
160
|
+
continue
|
|
161
|
+
total_lines = sum(max(c.lines_changed, 1) for _, c in matches)
|
|
162
|
+
for repo, commit in matches:
|
|
163
|
+
task = task_for(repo, commit, "fuzzy" if commit.ts <= rest[-1].ts + PAD else "grace")
|
|
164
|
+
share = max(commit.lines_changed, 1) / total_lines
|
|
165
|
+
_add_bursts(task, rest, share)
|
|
166
|
+
|
|
167
|
+
result.tasks = sorted(tasks.values(), key=lambda t: t.ts)
|
|
168
|
+
return result
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _add_bursts(task: Task, segment: list[Request], weight: float) -> None:
|
|
172
|
+
"""The final work burst before a commit is its direct cost; earlier bursts are lead-up.
|
|
173
|
+
|
|
174
|
+
r r r ··· 2h+ break ··· r r ··· 5h break ··· r r r [commit]
|
|
175
|
+
└ lead-up ┘ └ lead-up ┘ └ direct ┘
|
|
176
|
+
"""
|
|
177
|
+
start = 0
|
|
178
|
+
for k in range(1, len(segment)):
|
|
179
|
+
if segment[k].ts - segment[k - 1].ts >= BREAK:
|
|
180
|
+
start = k
|
|
181
|
+
for k, r in enumerate(segment):
|
|
182
|
+
task.add(r, weight, lead_up=k < start)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _load_repo(repo: str, since: float, now: float, refresh: bool) -> tuple[list[Commit], dict[str, str], RepoInfo]:
|
|
186
|
+
info = RepoInfo()
|
|
187
|
+
commits: list[Commit] = []
|
|
188
|
+
statuses: dict[str, str] = {}
|
|
189
|
+
try:
|
|
190
|
+
commits = load_commits(repo, since)
|
|
191
|
+
checker = KeptChecker(repo, since, now, refresh=refresh, commits=commits)
|
|
192
|
+
info.ref, info.ref_updated = checker.ref, checker.ref_updated
|
|
193
|
+
statuses = {sha: s.value for sha, s in checker.classify(commits).items()}
|
|
194
|
+
except GitError as exc:
|
|
195
|
+
# Commits may still have loaded (e.g. no default branch): attribute
|
|
196
|
+
# them, and their status reads "git error" in the receipt.
|
|
197
|
+
info.error = str(exc)
|
|
198
|
+
return commits, statuses, info
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _repo_relative(touched: set[tuple[str, str]], repo: str) -> set[str]:
|
|
202
|
+
out = set()
|
|
203
|
+
for cwd, path in touched:
|
|
204
|
+
full = os.path.normpath(path if os.path.isabs(path) else os.path.join(cwd, path))
|
|
205
|
+
full = os.path.realpath(full) if os.path.exists(full) else full
|
|
206
|
+
rel = os.path.relpath(full, repo)
|
|
207
|
+
if not rel.startswith(".."):
|
|
208
|
+
out.add(rel)
|
|
209
|
+
return out
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _fuzzy(
|
|
213
|
+
rest: list[Request],
|
|
214
|
+
touched: set[str],
|
|
215
|
+
commits: dict[str, Commit],
|
|
216
|
+
exact: set[tuple[str, str]],
|
|
217
|
+
repo: str,
|
|
218
|
+
email: str,
|
|
219
|
+
) -> list[Commit]:
|
|
220
|
+
"""Your own hand-made commits in the session's window (+2h grace) that touch files it edited.
|
|
221
|
+
|
|
222
|
+
Authorship matters: after a pull, teammates' merged commits land inside the
|
|
223
|
+
window and touch the same files. Without a configured user.email we don't guess.
|
|
224
|
+
"""
|
|
225
|
+
if not touched or not email:
|
|
226
|
+
return []
|
|
227
|
+
start, end = rest[0].ts - PAD, rest[-1].ts + GRACE
|
|
228
|
+
return sorted(
|
|
229
|
+
(
|
|
230
|
+
c
|
|
231
|
+
for c in commits.values()
|
|
232
|
+
if start <= c.ts <= end
|
|
233
|
+
and c.author_email == email
|
|
234
|
+
and (repo, c.sha) not in exact
|
|
235
|
+
and touched & set(c.files)
|
|
236
|
+
),
|
|
237
|
+
key=lambda c: c.ts,
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _anchors(calls: list[tuple[str, CommitCall]], commits_by_repo: dict[str, dict[str, Commit]]) -> list[tuple[str, Commit]]:
|
|
242
|
+
"""(repo, commit) whose time falls inside one of the session's commit calls, oldest first."""
|
|
243
|
+
found: dict[tuple[str, str], tuple[str, Commit]] = {}
|
|
244
|
+
for repo, call in calls:
|
|
245
|
+
start = call.start - PAD
|
|
246
|
+
end = (call.end if call.end is not None else call.start) + PAD
|
|
247
|
+
for commit in commits_by_repo.get(repo, {}).values():
|
|
248
|
+
if start <= commit.ts <= end:
|
|
249
|
+
found[(repo, commit.sha)] = (repo, commit)
|
|
250
|
+
return sorted(found.values(), key=lambda rc: rc[1].ts)
|
|
251
|
+
|
|
252
|
+
|
aimpg-0.1.0/aimpg/cli.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""`aimpg report`: print the fuel receipt for your local Claude Code usage."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
import time
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from aimpg.attribution import attribute
|
|
11
|
+
from aimpg.gitkept import DAY
|
|
12
|
+
from aimpg.logs import DEFAULT_ROOT, iter_log_files, parse_logs
|
|
13
|
+
from aimpg.receipt import render
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def main(argv: list[str] | None = None) -> int:
|
|
17
|
+
parser = argparse.ArgumentParser(prog="aimpg", description=__doc__)
|
|
18
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
19
|
+
report = sub.add_parser("report", help="print the energy receipt")
|
|
20
|
+
report.add_argument("--days", type=int, default=30, help="window size in days (default 30)")
|
|
21
|
+
report.add_argument("--logs", type=Path, default=DEFAULT_ROOT, help="Claude Code projects dir")
|
|
22
|
+
report.add_argument("--fetch", action="store_true", help="git fetch each repo first (uses the network)")
|
|
23
|
+
args = parser.parse_args(argv)
|
|
24
|
+
|
|
25
|
+
if args.days <= 0:
|
|
26
|
+
parser.error("--days must be positive")
|
|
27
|
+
if not args.logs.is_dir():
|
|
28
|
+
print(f"No Claude Code logs found at {args.logs}. Nothing to report.")
|
|
29
|
+
return 0
|
|
30
|
+
|
|
31
|
+
now = time.time()
|
|
32
|
+
since = now - args.days * DAY
|
|
33
|
+
parsed = parse_logs(iter_log_files(args.logs))
|
|
34
|
+
attribution = attribute(parsed, since, now, refresh=args.fetch)
|
|
35
|
+
sys.stdout.write(render(parsed, attribution, since, now))
|
|
36
|
+
return 0
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
if __name__ == "__main__":
|
|
40
|
+
raise SystemExit(main())
|