aimpg 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {aimpg-0.2.0 → aimpg-0.3.0}/PKG-INFO +26 -12
  2. {aimpg-0.2.0 → aimpg-0.3.0}/README.md +25 -11
  3. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/cli.py +35 -2
  4. aimpg-0.3.0/aimpg/codex_logs.py +142 -0
  5. aimpg-0.3.0/aimpg/durable.py +119 -0
  6. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/factors.json +35 -15
  7. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/gitkept.py +5 -0
  8. aimpg-0.3.0/aimpg/ledger.py +61 -0
  9. aimpg-0.3.0/aimpg/live.py +400 -0
  10. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/receipt.py +27 -2
  11. aimpg-0.3.0/docs/STRATEGY.md +66 -0
  12. {aimpg-0.2.0 → aimpg-0.3.0}/pyproject.toml +1 -1
  13. {aimpg-0.2.0 → aimpg-0.3.0}/tests/conftest.py +9 -0
  14. aimpg-0.3.0/tests/test_codex_logs.py +59 -0
  15. aimpg-0.3.0/tests/test_durable.py +77 -0
  16. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_energy.py +8 -0
  17. aimpg-0.3.0/tests/test_live.py +132 -0
  18. {aimpg-0.2.0 → aimpg-0.3.0}/uv.lock +1 -1
  19. {aimpg-0.2.0 → aimpg-0.3.0}/.github/workflows/publish.yml +0 -0
  20. {aimpg-0.2.0 → aimpg-0.3.0}/.github/workflows/test.yml +0 -0
  21. {aimpg-0.2.0 → aimpg-0.3.0}/.gitignore +0 -0
  22. {aimpg-0.2.0 → aimpg-0.3.0}/LICENSE +0 -0
  23. {aimpg-0.2.0 → aimpg-0.3.0}/TODOS.md +0 -0
  24. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/__init__.py +0 -0
  25. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/attribution.py +0 -0
  26. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/coach.py +0 -0
  27. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/cost.py +0 -0
  28. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/energy.py +0 -0
  29. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/equivalence.py +0 -0
  30. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/equivalences.json +0 -0
  31. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/logs.py +0 -0
  32. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/model.py +0 -0
  33. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/prices.json +0 -0
  34. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/__init__.py +0 -0
  35. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/cli.py +0 -0
  36. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/fake_agent.py +0 -0
  37. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/hints.py +0 -0
  38. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/picker.py +0 -0
  39. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/proxy.py +0 -0
  40. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/run.py +0 -0
  41. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/sandbox.py +0 -0
  42. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/select.py +0 -0
  43. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/setups.py +0 -0
  44. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/stats.py +0 -0
  45. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/replay/workspace.py +0 -0
  46. {aimpg-0.2.0 → aimpg-0.3.0}/aimpg/share.py +0 -0
  47. {aimpg-0.2.0 → aimpg-0.3.0}/docs/designs/aimpg-design.md +0 -0
  48. {aimpg-0.2.0 → aimpg-0.3.0}/evals/attribution_eval.py +0 -0
  49. {aimpg-0.2.0 → aimpg-0.3.0}/evals/label.py +0 -0
  50. {aimpg-0.2.0 → aimpg-0.3.0}/spikes/agent/RESULTS.md +0 -0
  51. {aimpg-0.2.0 → aimpg-0.3.0}/spikes/agent/allowlist_proxy.py +0 -0
  52. {aimpg-0.2.0 → aimpg-0.3.0}/spikes/agent/make_profile.py +0 -0
  53. {aimpg-0.2.0 → aimpg-0.3.0}/spikes/egress/RESULTS.md +0 -0
  54. {aimpg-0.2.0 → aimpg-0.3.0}/spikes/egress/allowlist_proxy.py +0 -0
  55. {aimpg-0.2.0 → aimpg-0.3.0}/spikes/egress/replay.sb +0 -0
  56. {aimpg-0.2.0 → aimpg-0.3.0}/tests/gitrepo.py +0 -0
  57. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_attribution.py +0 -0
  58. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_coach.py +0 -0
  59. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_cost_and_equivalence.py +0 -0
  60. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_eval_scoring.py +0 -0
  61. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_gitkept.py +0 -0
  62. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_logs.py +0 -0
  63. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_perf.py +0 -0
  64. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_replay_cli.py +0 -0
  65. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_replay_e2e.py +0 -0
  66. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_replay_hints.py +0 -0
  67. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_replay_picker.py +0 -0
  68. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_replay_proxy.py +0 -0
  69. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_replay_sandbox.py +0 -0
  70. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_replay_stats.py +0 -0
  71. {aimpg-0.2.0 → aimpg-0.3.0}/tests/test_share.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: aimpg
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Real-world energy per solved task for AI coding agents
5
5
  Project-URL: Homepage, https://github.com/kumarganduri/aimpg
6
6
  Project-URL: Issues, https://github.com/kumarganduri/aimpg/issues
@@ -28,28 +28,40 @@ uvx aimpg report
28
28
 
29
29
  ```
30
30
  YOUR AI CODING
31
- Energy 11.34 kWh – 92.22 kWh
32
- ≈ boiling a litre of water in a kettle 110–920 times · driving an electric car 67–540 km · 5.2–42.2 kg CO₂
33
- Money $1,210 API-equivalent (Anthropic's published prices)
34
- Output 328 commits made with AI help
31
+ Energy 14.46 kWh – 130.59 kWh
32
+ ≈ boiling a litre of water in a kettle 140–1,300 times · driving an electric car 85–770 km · 6.6–59.8 kg CO₂
33
+ Money $1,284 API-equivalent (Anthropic's published prices)
34
+ Output 353 commits made with AI help
35
35
 
36
- A typical kept commit: 37.4 Wh – 283.5 Wh ≈ 2–17 full phone charges · $3.62
36
+ A typical kept commit: 36.6 Wh – 313.0 Wh ≈ 2–18 full phone charges · $2.05
37
37
 
38
38
  WHAT WOULD HAVE SAVED THE MOST (measured on your logs; upper bounds that overlap)
39
- 1. Start a fresh session after each commit: up to 44% less energy, $367
40
- 44% of your AI energy went to re-reading conversation from before your last commit.
39
+ 1. Start a fresh session after each commit: up to 45% less energy, $397
40
+ 45% of your AI energy went to re-reading conversation from before your last commit.
41
41
  → After committing, start a new session (or /clear) for the next task.
42
- 2. Use a mid-size model for routine work: up to 16% less energy, $316
42
+ 2. Use a mid-size model for routine work: up to 16% less energy, $333
43
43
  94% of your AI energy ran on the largest models (Opus/Fable class).
44
44
  → Check a cheaper model is good enough on your own commits: `aimpg replay models`.
45
+
46
+ WORKING CHANGES (commits that shipped and lasted 30 days)
47
+ Available on 2026-10-22: aimpg is keeping your history so it can tell.
45
48
  ```
46
49
 
47
50
  *(Real output from the author's last 30 days.)*
48
51
 
49
52
  ## What you can do with it
50
53
 
54
+ ### 0. See it live while you work
55
+ ```bash
56
+ uv tool install aimpg && aimpg live install # shows the change, asks y/N, keeps a backup
57
+ ```
58
+ ```
59
+ ⚡ task $1.96 · usual $2.05 · ctx 160k (92% from before your last commit) · /clear ≈ -92% per request · 5h 63%
60
+ ```
61
+ After the agent commits, one line for you only (never sent to the model): *"that commit cost $2.40 over 31 AI requests. 92% of the context is from earlier work: /clear before the next task saves ~92% per request."* Undo with `aimpg live uninstall`.
62
+
51
63
  ### 1. See your AI footprint in terms you can picture
52
- `aimpg report` reads the Claude Code logs already on your machine and ties every AI request to the git commit it produced. Energy is shown as an honest range and translated into kettles, phone charges, EV kilometres and CO₂. Money is the API-equivalent cost at Anthropic's published prices (for subscribers, what the same work would cost on the API).
64
+ `aimpg report` reads the Claude Code (and Codex CLI) logs already on your machine and ties every AI request to the git commit it produced. Energy is shown as an honest range and translated into kettles, phone charges, EV kilometres and CO₂. Money is the API-equivalent cost at Anthropic's published prices (for subscribers, what the same work would cost on the API).
53
65
 
54
66
  ### 2. Get tips measured on your own habits, not generic advice
55
67
  The receipt replays your own logs under "what if" rules and tells you what would have saved the most, in Wh and dollars. For example: how much energy went into re-reading old conversation in long sessions, or into running the biggest model for routine fixes.
@@ -99,13 +111,15 @@ One row per AI-assisted commit: date, repo, energy range, CO₂ range and cost.
99
111
 
100
112
  **Matching requests to commits:** when the agent runs `git commit`, the commit lands inside that tool call, so the match is exact. Within a session, requests since the previous commit belong to the next one. Work before a 2h+ break is shown separately as *lead-up*. Hand-made commits are matched only if you authored them and they touch files the session edited. On the author's history, a hand-labeled check matched 20/20 commits correctly (`evals/`).
101
113
 
102
- **Energy:** a physical formula, not a price proxy. Prefill compute for new input, one KV-cache re-read per output token (so long contexts cost more), plus datacenter overhead. Model sizes aren't public, so every number is a range. Sources for every factor are in [`aimpg/factors.json`](aimpg/factors.json), [`aimpg/prices.json`](aimpg/prices.json) and [`aimpg/equivalences.json`](aimpg/equivalences.json).
114
+ **Working changes:** aimpg keeps a small history of what each commit cost (never prompts or code), because Claude Code deletes its own logs after 30 days. Once a commit is 30 days old, it is judged: still on the main branch and not reverted or largely rewritten counts as a *working change*. The receipt then shows **cost per working change** and a **waste ratio** (AI spend on work that didn't last).
115
+
116
+ **Energy:** a physical formula, not a price proxy. Prefill compute for new input, one KV-cache re-read per output token (so long contexts cost more), plus datacenter overhead. Model sizes aren't public, so every number is a range. Overheads follow Google's full-stack measurement of a median Gemini prompt (chips, host CPU and memory, idle capacity, cooling ≈ 1.7× the chips alone), and a chat-sized prompt on a mid-size model comes out at 0.017–0.34 Wh, bracketing Google's disclosed 0.24 Wh ([arXiv 2508.15734](https://arxiv.org/pdf/2508.15734)). Sources for every factor are in [`aimpg/factors.json`](aimpg/factors.json), [`aimpg/prices.json`](aimpg/prices.json) and [`aimpg/equivalences.json`](aimpg/equivalences.json).
103
117
 
104
118
  **Replays:** each run starts from the code just before your commit, in a fresh, history-free copy, inside a macOS sandbox. The only network allowed is the model API, through an allowlisting proxy. Your own tests judge the result, and edited tests are always restored before judging. Every token count is cross-checked against Claude Code's own totals. Details are in [docs/designs/aimpg-design.md](docs/designs/aimpg-design.md).
105
119
 
106
120
  ## Limits (honest list)
107
121
 
108
- - Claude Code logs only, for now. Replays need macOS and an Anthropic API key (about $0.10–0.30 per run).
122
+ - Claude Code and Codex CLI logs. Cursor keeps token usage on its servers, not on your machine, so it isn't covered yet. Codex's model prices aren't in the table yet; its requests count toward energy but are listed as unpriced. Replays need macOS and an Anthropic API key (about $0.10–0.30 per run).
109
123
  - Energy is an estimate with a wide range; dollars are close (within about 7% of Claude Code's own session totals on the author's logs).
110
124
  - Tips are upper bounds and overlap; they can't be added together.
111
125
  - Replays from commit messages alone are hard (12% solved on the author's repo); `--task-mode tests` shows the agent the tests, which makes tasks easier than real work but keeps comparisons fair.
@@ -8,28 +8,40 @@ uvx aimpg report
8
8
 
9
9
  ```
10
10
  YOUR AI CODING
11
- Energy 11.34 kWh – 92.22 kWh
12
- ≈ boiling a litre of water in a kettle 110–920 times · driving an electric car 67–540 km · 5.2–42.2 kg CO₂
13
- Money $1,210 API-equivalent (Anthropic's published prices)
14
- Output 328 commits made with AI help
11
+ Energy 14.46 kWh – 130.59 kWh
12
+ ≈ boiling a litre of water in a kettle 140–1,300 times · driving an electric car 85–770 km · 6.6–59.8 kg CO₂
13
+ Money $1,284 API-equivalent (Anthropic's published prices)
14
+ Output 353 commits made with AI help
15
15
 
16
- A typical kept commit: 37.4 Wh – 283.5 Wh ≈ 2–17 full phone charges · $3.62
16
+ A typical kept commit: 36.6 Wh – 313.0 Wh ≈ 2–18 full phone charges · $2.05
17
17
 
18
18
  WHAT WOULD HAVE SAVED THE MOST (measured on your logs; upper bounds that overlap)
19
- 1. Start a fresh session after each commit: up to 44% less energy, $367
20
- 44% of your AI energy went to re-reading conversation from before your last commit.
19
+ 1. Start a fresh session after each commit: up to 45% less energy, $397
20
+ 45% of your AI energy went to re-reading conversation from before your last commit.
21
21
  → After committing, start a new session (or /clear) for the next task.
22
- 2. Use a mid-size model for routine work: up to 16% less energy, $316
22
+ 2. Use a mid-size model for routine work: up to 16% less energy, $333
23
23
  94% of your AI energy ran on the largest models (Opus/Fable class).
24
24
  → Check a cheaper model is good enough on your own commits: `aimpg replay models`.
25
+
26
+ WORKING CHANGES (commits that shipped and lasted 30 days)
27
+ Available on 2026-10-22: aimpg is keeping your history so it can tell.
25
28
  ```
26
29
 
27
30
  *(Real output from the author's last 30 days.)*
28
31
 
29
32
  ## What you can do with it
30
33
 
34
+ ### 0. See it live while you work
35
+ ```bash
36
+ uv tool install aimpg && aimpg live install # shows the change, asks y/N, keeps a backup
37
+ ```
38
+ ```
39
+ ⚡ task $1.96 · usual $2.05 · ctx 160k (92% from before your last commit) · /clear ≈ -92% per request · 5h 63%
40
+ ```
41
+ After the agent commits, one line for you only (never sent to the model): *"that commit cost $2.40 over 31 AI requests. 92% of the context is from earlier work: /clear before the next task saves ~92% per request."* Undo with `aimpg live uninstall`.
42
+
31
43
  ### 1. See your AI footprint in terms you can picture
32
- `aimpg report` reads the Claude Code logs already on your machine and ties every AI request to the git commit it produced. Energy is shown as an honest range and translated into kettles, phone charges, EV kilometres and CO₂. Money is the API-equivalent cost at Anthropic's published prices (for subscribers, what the same work would cost on the API).
44
+ `aimpg report` reads the Claude Code (and Codex CLI) logs already on your machine and ties every AI request to the git commit it produced. Energy is shown as an honest range and translated into kettles, phone charges, EV kilometres and CO₂. Money is the API-equivalent cost at Anthropic's published prices (for subscribers, what the same work would cost on the API).
33
45
 
34
46
  ### 2. Get tips measured on your own habits, not generic advice
35
47
  The receipt replays your own logs under "what if" rules and tells you what would have saved the most, in Wh and dollars. For example: how much energy went into re-reading old conversation in long sessions, or into running the biggest model for routine fixes.
@@ -79,13 +91,15 @@ One row per AI-assisted commit: date, repo, energy range, CO₂ range and cost.
79
91
 
80
92
  **Matching requests to commits:** when the agent runs `git commit`, the commit lands inside that tool call, so the match is exact. Within a session, requests since the previous commit belong to the next one. Work before a 2h+ break is shown separately as *lead-up*. Hand-made commits are matched only if you authored them and they touch files the session edited. On the author's history, a hand-labeled check matched 20/20 commits correctly (`evals/`).
81
93
 
82
- **Energy:** a physical formula, not a price proxy. Prefill compute for new input, one KV-cache re-read per output token (so long contexts cost more), plus datacenter overhead. Model sizes aren't public, so every number is a range. Sources for every factor are in [`aimpg/factors.json`](aimpg/factors.json), [`aimpg/prices.json`](aimpg/prices.json) and [`aimpg/equivalences.json`](aimpg/equivalences.json).
94
+ **Working changes:** aimpg keeps a small history of what each commit cost (never prompts or code), because Claude Code deletes its own logs after 30 days. Once a commit is 30 days old, it is judged: still on the main branch and not reverted or largely rewritten counts as a *working change*. The receipt then shows **cost per working change** and a **waste ratio** (AI spend on work that didn't last).
95
+
96
+ **Energy:** a physical formula, not a price proxy. Prefill compute for new input, one KV-cache re-read per output token (so long contexts cost more), plus datacenter overhead. Model sizes aren't public, so every number is a range. Overheads follow Google's full-stack measurement of a median Gemini prompt (chips, host CPU and memory, idle capacity, cooling ≈ 1.7× the chips alone), and a chat-sized prompt on a mid-size model comes out at 0.017–0.34 Wh, bracketing Google's disclosed 0.24 Wh ([arXiv 2508.15734](https://arxiv.org/pdf/2508.15734)). Sources for every factor are in [`aimpg/factors.json`](aimpg/factors.json), [`aimpg/prices.json`](aimpg/prices.json) and [`aimpg/equivalences.json`](aimpg/equivalences.json).
83
97
 
84
98
  **Replays:** each run starts from the code just before your commit, in a fresh, history-free copy, inside a macOS sandbox. The only network allowed is the model API, through an allowlisting proxy. Your own tests judge the result, and edited tests are always restored before judging. Every token count is cross-checked against Claude Code's own totals. Details are in [docs/designs/aimpg-design.md](docs/designs/aimpg-design.md).
85
99
 
86
100
  ## Limits (honest list)
87
101
 
88
- - Claude Code logs only, for now. Replays need macOS and an Anthropic API key (about $0.10–0.30 per run).
102
+ - Claude Code and Codex CLI logs. Cursor keeps token usage on its servers, not on your machine, so it isn't covered yet. Codex's model prices aren't in the table yet; its requests count toward energy but are listed as unpriced. Replays need macOS and an Anthropic API key (about $0.10–0.30 per run).
89
103
  - Energy is an estimate with a wide range; dollars are close (within about 7% of Claude Code's own session totals on the author's logs).
90
104
  - Tips are upper bounds and overlap; they can't be added together.
91
105
  - Replays from commit messages alone are hard (12% solved on the author's repo); `--task-mode tests` shows the agent the tests, which makes tasks easier than real work but keeps comparisons fair.
@@ -4,6 +4,7 @@
4
4
  pr AI energy + cost of the current branch's commits (markdown; --post to the PR)
5
5
  export one CSV row per AI-assisted commit, for teams and sustainability reports
6
6
  replay compare agent setups and models on your own past commits
7
+ live install/uninstall the live status line + post-commit note in Claude Code
7
8
  """
8
9
 
9
10
  from __future__ import annotations
@@ -24,13 +25,16 @@ from aimpg.replay import cli as replay_cli
24
25
  def _common(p: argparse.ArgumentParser, days: int) -> None:
25
26
  p.add_argument("--days", type=int, default=days, help=f"window size in days (default {days})")
26
27
  p.add_argument("--logs", type=Path, default=DEFAULT_ROOT, help="Claude Code projects dir")
28
+ p.add_argument("--codex-logs", type=Path, default=None, help="Codex sessions dir (default ~/.codex/sessions)")
27
29
  p.add_argument("--fetch", action="store_true", help="git fetch each repo first (uses the network)")
28
30
 
29
31
 
30
32
  def _analyze(args):
33
+ from aimpg import codex_logs
34
+
31
35
  now = time.time()
32
36
  since = now - args.days * DAY
33
- parsed = parse_logs(iter_log_files(args.logs))
37
+ parsed = codex_logs.merge(parse_logs(iter_log_files(args.logs)), codex_logs.parse_codex(codex_logs.iter_files(args.codex_logs)))
34
38
  return parsed, attribute(parsed, since, now, refresh=args.fetch), since, now
35
39
 
36
40
 
@@ -51,10 +55,34 @@ def main(argv: list[str] | None = None) -> int:
51
55
  export.add_argument("--no-subjects", action="store_true", help="leave commit messages out (privacy)")
52
56
  export.add_argument("--no-authors", action="store_true", help="leave author emails out (privacy)")
53
57
 
58
+ live = sub.add_parser("live", help="live status line + post-commit note inside Claude Code")
59
+ live.add_argument("action", choices=("install", "uninstall"))
60
+ sub.add_parser("statusline", help=argparse.SUPPRESS) # called by Claude Code
61
+ hook = sub.add_parser("hook", help=argparse.SUPPRESS) # called by Claude Code
62
+ hook.add_argument("event", choices=("post-commit",))
63
+
54
64
  replay_cli.add_parser(sub)
55
65
  args = parser.parse_args(argv)
56
66
  if args.command == "replay":
57
67
  return replay_cli.main(args)
68
+ if args.command in ("statusline", "hook", "live"):
69
+ from aimpg import live as live_mod
70
+
71
+ if args.command == "statusline":
72
+ return live_mod.run_statusline(live_mod.previous_statusline())
73
+ if args.command == "hook":
74
+ return live_mod.run_post_commit_hook()
75
+ if args.action == "install":
76
+ import shutil
77
+
78
+ exe = shutil.which("aimpg")
79
+ if exe is None:
80
+ print("Install aimpg permanently first so Claude Code can call it: uv tool install aimpg")
81
+ return 1
82
+ print(live_mod.install(exe=exe))
83
+ else:
84
+ print(live_mod.uninstall())
85
+ return 0
58
86
 
59
87
  if args.days <= 0:
60
88
  parser.error("--days must be positive")
@@ -85,7 +113,12 @@ def main(argv: list[str] | None = None) -> int:
85
113
  print("\nPosted to the pull request.")
86
114
  return 0
87
115
 
88
- sys.stdout.write(render(parsed, attribution, since, now))
116
+ from aimpg import durable, ledger
117
+ from aimpg.receipt import task_energy
118
+
119
+ ledger.record(task_energy(attribution.tasks))
120
+ judged = durable.judge(ledger.load(), now)
121
+ sys.stdout.write(render(parsed, attribution, since, now, judged))
89
122
  return 0
90
123
 
91
124
 
@@ -0,0 +1,142 @@
1
+ """Read OpenAI Codex CLI session logs (~/.codex/sessions/**/rollout-*.jsonl).
2
+
3
+ Verified on real files (codex-cli 0.159):
4
+ * `event_msg` / `token_count` events carry `info.last_token_usage` for each
5
+ model call: input_tokens (includes cached), cached_input_tokens,
6
+ cache_write_input_tokens, output_tokens, reasoning_output_tokens.
7
+ * `turn_context` carries the model (e.g. gpt-6-luna) and cwd; `session_meta`
8
+ carries the session id and starting cwd.
9
+
10
+ Not yet verified on a real file (no local session ran shell commands): how a
11
+ `git commit` tool call is recorded. Detection is tolerant: any response_item
12
+ tool call whose arguments contain `git ... commit`, closed by an output item
13
+ with the same call_id. Commits made by hand are still matched by the fuzzy tier.
14
+
15
+ Codex token usage is mapped onto the same buckets as Claude Code:
16
+ fresh_in = input − cached − cache_write · cache_read = cached
17
+ cache_write = cache_write_input · output = output + reasoning (generated text)
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import json
23
+ from pathlib import Path
24
+ from typing import Iterable, Iterator
25
+
26
+ from aimpg.logs import _GIT_COMMIT, commit_cwd, parse_ts
27
+ from aimpg.model import CommitCall, ParseResult, Request, Usage
28
+
29
+ DEFAULT_ROOT = Path.home() / ".codex" / "sessions"
30
+ TOOL_CALLS = ("function_call", "custom_tool_call", "local_shell_call")
31
+ TOOL_OUTPUTS = ("function_call_output", "custom_tool_call_output", "local_shell_call_output")
32
+
33
+
34
+ def iter_files(root: Path | None = None) -> Iterator[Path]:
35
+ root = root or DEFAULT_ROOT # looked up at call time so tests can point it elsewhere
36
+ if root.is_dir():
37
+ yield from sorted(root.rglob("*.jsonl"))
38
+
39
+
40
+ def _command_text(payload: dict) -> str:
41
+ for key in ("arguments", "input", "action"):
42
+ value = payload.get(key)
43
+ if isinstance(value, str):
44
+ try:
45
+ value = json.loads(value)
46
+ except ValueError:
47
+ return value
48
+ if isinstance(value, dict):
49
+ cmd = value.get("command") or value.get("cmd")
50
+ if isinstance(cmd, list):
51
+ return " ".join(str(c) for c in cmd)
52
+ if cmd:
53
+ return str(cmd)
54
+ return ""
55
+
56
+
57
+ def parse_codex(files: Iterable[Path]) -> ParseResult:
58
+ result = ParseResult()
59
+ stats = {"files": 0, "rows": 0, "unique_requests": 0, "corrupt_rows": 0}
60
+ versions: dict[str, int] = {}
61
+ for path in files:
62
+ stats["files"] += 1
63
+ session, cwd, model = path.stem, "", ""
64
+ pending: dict[str, tuple[float, str]] = {}
65
+ n = 0
66
+ with open(path, "rb") as fh:
67
+ for line in fh:
68
+ if not line.strip():
69
+ continue
70
+ stats["rows"] += 1
71
+ try:
72
+ row = json.loads(line)
73
+ except ValueError:
74
+ stats["corrupt_rows"] += 1
75
+ continue
76
+ kind, payload = row.get("type"), row.get("payload") or {}
77
+ ts = parse_ts(row.get("timestamp"))
78
+ if kind == "session_meta":
79
+ session = str(payload.get("session_id") or payload.get("id") or session)
80
+ cwd = str(payload.get("cwd") or cwd)
81
+ if payload.get("cli_version"):
82
+ versions["codex " + str(payload["cli_version"])] = versions.get("codex " + str(payload["cli_version"]), 0) + 1
83
+ elif kind == "turn_context":
84
+ model = str(payload.get("model") or model)
85
+ cwd = str(payload.get("cwd") or cwd)
86
+ elif kind == "event_msg" and payload.get("type") == "token_count" and ts is not None:
87
+ usage = ((payload.get("info") or {}).get("last_token_usage")) or {}
88
+ if not usage:
89
+ continue
90
+ cached = int(usage.get("cached_input_tokens") or 0)
91
+ written = int(usage.get("cache_write_input_tokens") or 0)
92
+ total_in = int(usage.get("input_tokens") or 0)
93
+ n += 1
94
+ result.requests.append(Request(
95
+ id=f"codex:{session}:{n}",
96
+ session_id=f"codex:{session}",
97
+ model=model or "codex-unknown",
98
+ ts=ts,
99
+ usage=Usage(
100
+ fresh_in=max(0, total_in - cached - written),
101
+ cache_write=written,
102
+ cache_read=cached,
103
+ output=int(usage.get("output_tokens") or 0) + int(usage.get("reasoning_output_tokens") or 0),
104
+ ),
105
+ cwd=cwd,
106
+ ))
107
+ elif kind == "response_item" and ts is not None:
108
+ ptype = payload.get("type")
109
+ if ptype in TOOL_CALLS:
110
+ command = _command_text(payload)
111
+ if _GIT_COMMIT.search(command):
112
+ repo_dir = commit_cwd(command, cwd) or cwd
113
+ pending[str(payload.get("call_id") or payload.get("id"))] = (ts, repo_dir)
114
+ elif ptype in TOOL_OUTPUTS:
115
+ call = pending.pop(str(payload.get("call_id")), None)
116
+ if call:
117
+ result.commit_calls.append(CommitCall(f"codex:{session}", call[1], call[0], ts))
118
+ stats["unique_requests"] += n
119
+ result.requests.sort(key=lambda r: r.ts)
120
+ result.commit_calls.sort(key=lambda c: c.start)
121
+ result.stats = stats
122
+ result.versions = versions
123
+ return result
124
+
125
+
126
+ def merge(*parts: ParseResult) -> ParseResult:
127
+ """One ParseResult from several agents' logs."""
128
+ out = ParseResult()
129
+ stats: dict[str, int] = {}
130
+ for p in parts:
131
+ out.requests.extend(p.requests)
132
+ out.commit_calls.extend(p.commit_calls)
133
+ for k, v in p.files_touched.items():
134
+ out.files_touched.setdefault(k, set()).update(v)
135
+ for k, v in p.stats.items():
136
+ stats[k] = stats.get(k, 0) + v
137
+ for k, v in p.versions.items():
138
+ out.versions[k] = out.versions.get(k, 0) + v
139
+ out.requests.sort(key=lambda r: r.ts)
140
+ out.commit_calls.sort(key=lambda c: c.start)
141
+ out.stats = stats
142
+ return out
@@ -0,0 +1,119 @@
1
+ """Working (durable) changes: AI-assisted commits that shipped and stayed.
2
+
3
+ A commit becomes judgeable 30 days after it was made. It is:
4
+
5
+ reverted a later commit says "This reverts commit <sha>"
6
+ reworked within 30 days, later commits on the default branch deleted at
7
+ least 30% of the meaningful lines it added
8
+ not kept it never reached the default branch (or is unknown)
9
+ durable none of the above
10
+
11
+ cost per working change = AI $ on judged commits / number of durable ones
12
+ waste ratio = AI $ on judged commits that weren't durable / AI $ on judged commits
13
+
14
+ Commits younger than 30 days are "too new" and left out of both numbers.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import subprocess
20
+ import time
21
+ from collections import defaultdict
22
+ from dataclasses import dataclass, field
23
+ from datetime import datetime, timezone
24
+
25
+ from aimpg.gitkept import DAY, GitError, KeptChecker, load_commits
26
+
27
+ MATURE_AFTER = 30 * DAY
28
+ REWORK_SHARE = 0.30
29
+
30
+
31
+ @dataclass
32
+ class Judged:
33
+ durable: list[dict] = field(default_factory=list)
34
+ reverted: list[dict] = field(default_factory=list)
35
+ reworked: list[dict] = field(default_factory=list)
36
+ not_kept: list[dict] = field(default_factory=list)
37
+ too_new: list[dict] = field(default_factory=list)
38
+ errors: dict[str, str] = field(default_factory=dict)
39
+
40
+ @property
41
+ def judged(self) -> list[dict]:
42
+ return self.durable + self.reverted + self.reworked + self.not_kept
43
+
44
+ def cost_per_durable(self) -> float | None:
45
+ usd = sum(r["usd"] for r in self.judged)
46
+ return usd / len(self.durable) if self.durable else None
47
+
48
+ def waste_ratio(self) -> float | None:
49
+ usd = sum(r["usd"] for r in self.judged)
50
+ wasted = usd - sum(r["usd"] for r in self.durable)
51
+ return wasted / usd if usd else None
52
+
53
+ def first_judgeable(self) -> float | None:
54
+ return min((r["ts"] for r in self.too_new), default=None)
55
+
56
+
57
+ def reverted_shas(repo: str, since: float) -> set[str]:
58
+ proc = subprocess.run(
59
+ ["git", "-C", repo, "log", "--all", f"--since={int(since)}", "--grep=This reverts commit", "--format=%B%x1e"],
60
+ capture_output=True, text=True, errors="replace",
61
+ )
62
+ out = set()
63
+ for body in proc.stdout.split("\x1e"):
64
+ for word in body.replace(".", " ").split():
65
+ if len(word) == 40 and all(c in "0123456789abcdef" for c in word):
66
+ out.add(word)
67
+ return out
68
+
69
+
70
+ def judge(ledger: dict[str, dict], now: float | None = None) -> Judged:
71
+ now = now or time.time()
72
+ result = Judged()
73
+ by_repo: dict[str, list[dict]] = defaultdict(list)
74
+ for rec in ledger.values():
75
+ (by_repo[rec["repo"]] if now - rec["ts"] >= MATURE_AFTER else result.too_new).append(rec)
76
+ for repo, recs in by_repo.items():
77
+ oldest = min(r["ts"] for r in recs)
78
+ try:
79
+ checker = KeptChecker(repo, oldest - DAY, now)
80
+ commits = load_commits(repo, oldest - DAY, reflog=False)
81
+ later = load_commits(repo, oldest, refs=[checker.ref])
82
+ statuses = checker.classify(commits)
83
+ reverts = reverted_shas(repo, oldest)
84
+ except GitError as exc:
85
+ result.errors[repo] = str(exc)
86
+ result.not_kept.extend(recs)
87
+ continue
88
+ own = {c.sha: c for c in commits}
89
+ for rec in recs:
90
+ status = statuses.get(rec["sha"])
91
+ if rec["sha"] in reverts:
92
+ result.reverted.append(rec)
93
+ elif status is None or not status.is_kept:
94
+ result.not_kept.append(rec)
95
+ elif _reworked(own.get(rec["sha"]), later):
96
+ result.reworked.append(rec)
97
+ else:
98
+ result.durable.append(rec)
99
+ return result
100
+
101
+
102
+ def _reworked(commit, later) -> bool:
103
+ if commit is None:
104
+ return False
105
+ added = {p: lines for p, lines in commit.added.items() if lines}
106
+ total = sum(len(v) for v in added.values())
107
+ if total == 0:
108
+ return False
109
+ gone = set()
110
+ for c in later:
111
+ if c.sha == commit.sha or not (commit.ts < c.ts <= commit.ts + MATURE_AFTER):
112
+ continue
113
+ for path, lines in added.items():
114
+ gone |= {(path, l) for l in lines & c.removed.get(path, set())}
115
+ return len(gone) / total >= REWORK_SHARE
116
+
117
+
118
+ def day(ts: float) -> str:
119
+ return datetime.fromtimestamp(ts, tz=timezone.utc).strftime("%Y-%m-%d")
@@ -1,20 +1,20 @@
1
1
  {
2
- "version": "2026-10-01",
2
+ "version": "2026-10-03b",
3
3
  "constants": {
4
4
  "joules_per_flop": {
5
- "value": 0.52e-12,
5
+ "value": 5.2e-13,
6
6
  "citation": "From Tokens to Watt-hours (arXiv 2607.26571): tensor-core coefficient alpha_TC = 0.52 pJ/FLOP, H100 BF16"
7
7
  },
8
8
  "joules_per_hbm_bit": {
9
- "value": 11.68e-12,
9
+ "value": 1.168e-11,
10
10
  "citation": "From Tokens to Watt-hours (arXiv 2607.26571): HBM energy e_HBM = 11.68 pJ/bit, H100"
11
11
  }
12
12
  },
13
13
  "ranges": {
14
14
  "overhead": {
15
- "low": 1.25,
16
- "high": 1.5,
17
- "citation": "EcoLogits LLM inference methodology: non-GPU server power 1.2 kW per 8-GPU server, provider PUE 1.09-1.20"
15
+ "low": 1.5,
16
+ "high": 2.0,
17
+ "citation": "Full-stack boundary per Google's measured breakdown of a median Gemini prompt (0.24 Wh = accelerators 58%, host CPU/DRAM 25%, idle provisioned machines 10%, PUE overhead 8%), i.e. total \u2248 1.7x accelerator energy: https://arxiv.org/pdf/2508.15734 . Range 1.5-2.0 brackets leaner and less efficient fleets. Previously 1.25-1.5 (PUE + server only), which left out host and idle energy."
18
18
  },
19
19
  "batch_size": {
20
20
  "low": 128,
@@ -27,8 +27,8 @@
27
27
  "citation": "Assumption: FP8 to BF16 serving precision"
28
28
  },
29
29
  "kv_bits_per_token": {
30
- "low": 0.5e6,
31
- "high": 3.0e6,
30
+ "low": 500000.0,
31
+ "high": 3000000.0,
32
32
  "citation": "Assumption: 2 x layers x kv_heads x head_dim x bytes for GQA models (e.g. 80x8x128x2x2 bytes = 2.6 Mbit BF16); FP8/compressed caches are lower"
33
33
  },
34
34
  "prefill_inefficiency": {
@@ -44,20 +44,40 @@
44
44
  },
45
45
  "classes": {
46
46
  "large": {
47
- "active_params_billion": {"low": 100, "high": 400},
48
- "matches": ["opus", "fable"],
47
+ "active_params_billion": {
48
+ "low": 100,
49
+ "high": 400
50
+ },
51
+ "matches": [
52
+ "opus",
53
+ "fable"
54
+ ],
49
55
  "citation": "Assumption: parameter counts are not disclosed for frontier closed models"
50
56
  },
51
57
  "mid": {
52
- "active_params_billion": {"low": 30, "high": 150},
53
- "matches": ["sonnet"],
58
+ "active_params_billion": {
59
+ "low": 30,
60
+ "high": 150
61
+ },
62
+ "matches": [
63
+ "sonnet"
64
+ ],
54
65
  "citation": "Assumption: parameter counts are not disclosed for frontier closed models"
55
66
  },
56
67
  "small": {
57
- "active_params_billion": {"low": 8, "high": 40},
58
- "matches": ["haiku"],
68
+ "active_params_billion": {
69
+ "low": 8,
70
+ "high": 40
71
+ },
72
+ "matches": [
73
+ "haiku"
74
+ ],
59
75
  "citation": "Assumption: parameter counts are not disclosed for frontier closed models"
60
76
  }
61
77
  },
62
- "default_class": "mid"
78
+ "default_class": "mid",
79
+ "calibration": {
80
+ "check": "A chat-sized prompt (600 input, 400 output tokens, no cache) on a mid-size model should bracket Google's disclosed 0.24 Wh median Gemini prompt (same full-stack boundary).",
81
+ "citation": "https://arxiv.org/pdf/2508.15734"
82
+ }
63
83
  }
@@ -54,6 +54,8 @@ class Commit:
54
54
  author_email: str = ""
55
55
  files: dict[str, tuple[int, int]] = field(default_factory=dict) # path -> (added, deleted)
56
56
  added: dict[str, set[str]] = field(default_factory=dict) # path -> meaningful added lines
57
+ removed: dict[str, set[str]] = field(default_factory=dict) # path -> meaningful removed lines (rework detection)
58
+ message: str = "" # full message body (revert detection); filled by load_commits
57
59
 
58
60
  @property
59
61
  def lines_changed(self) -> int:
@@ -182,6 +184,9 @@ def _parse_patch_log(text: str) -> list[Commit]:
182
184
  elif line.startswith("-"):
183
185
  a, d = commit.files[current]
184
186
  commit.files[current] = (a, d + 1)
187
+ stripped = line[1:].strip()
188
+ if len(stripped) >= _MIN_LINE:
189
+ commit.removed.setdefault(current, set()).add(stripped)
185
190
  commits.append(commit)
186
191
  return commits
187
192
 
@@ -0,0 +1,61 @@
1
+ """aimpg's own history: what each AI-assisted commit cost, kept beyond Claude Code's logs.
2
+
3
+ Claude Code deletes transcripts after 30 days by default, but a commit can only
4
+ be called durable after 30 days. So every report records a small line per
5
+ commit in ~/.aimpg/ledger.json: cost, energy range, request count. Never
6
+ prompts, code or diffs. Re-running is safe: a commit's record is replaced
7
+ only by one with at least as many requests, so deleted logs never shrink it.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import time
14
+ from pathlib import Path
15
+
16
+ LEDGER = Path.home() / ".aimpg" / "ledger.json"
17
+
18
+
19
+ def load(path: Path = LEDGER) -> dict[str, dict]:
20
+ try:
21
+ return json.loads(path.read_text())
22
+ except (OSError, ValueError):
23
+ return {}
24
+
25
+
26
+ def key(repo: str, sha: str) -> str:
27
+ return f"{repo}|{sha}"
28
+
29
+
30
+ def record(energies, path: Path = LEDGER) -> int:
31
+ """Upsert one line per commit from receipt.task_energy(...). Returns how many changed."""
32
+ data = load(path)
33
+ changed = 0
34
+ now = time.time()
35
+ for e in energies:
36
+ t = e.task
37
+ requests = len(t.requests) + len(t.lead_up)
38
+ if requests == 0:
39
+ continue
40
+ k = key(t.repo, t.sha)
41
+ old = data.get(k)
42
+ if old and old.get("requests", 0) > requests:
43
+ continue # logs were cleaned up since: keep the fuller record
44
+ data[k] = {
45
+ "repo": t.repo,
46
+ "sha": t.sha,
47
+ "ts": t.ts,
48
+ "subject": t.subject,
49
+ "author": t.author,
50
+ "requests": requests,
51
+ "usd": round(e.usd + e.lead_up_usd, 4),
52
+ "wh_low": round(e.total.low, 3),
53
+ "wh_high": round(e.total.high, 3),
54
+ "first_seen": (old or {}).get("first_seen", now),
55
+ }
56
+ changed += old != data[k]
57
+ path.parent.mkdir(parents=True, exist_ok=True)
58
+ tmp = path.with_suffix(".json.tmp")
59
+ tmp.write_text(json.dumps(data))
60
+ tmp.replace(path)
61
+ return changed