aimpg 0.4.2__tar.gz → 0.4.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. {aimpg-0.4.2 → aimpg-0.4.3}/PKG-INFO +3 -1
  2. {aimpg-0.4.2 → aimpg-0.4.3}/README.md +2 -0
  3. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/cli.py +4 -0
  4. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/verify.py +11 -1
  5. aimpg-0.4.3/aimpg/scoreboard.py +621 -0
  6. {aimpg-0.4.2 → aimpg-0.4.3}/pyproject.toml +1 -1
  7. {aimpg-0.4.2 → aimpg-0.4.3}/tests/conftest.py +3 -0
  8. aimpg-0.4.3/tests/test_scoreboard.py +201 -0
  9. {aimpg-0.4.2 → aimpg-0.4.3}/uv.lock +1 -1
  10. {aimpg-0.4.2 → aimpg-0.4.3}/.github/workflows/publish.yml +0 -0
  11. {aimpg-0.4.2 → aimpg-0.4.3}/.github/workflows/test.yml +0 -0
  12. {aimpg-0.4.2 → aimpg-0.4.3}/.gitignore +0 -0
  13. {aimpg-0.4.2 → aimpg-0.4.3}/LICENSE +0 -0
  14. {aimpg-0.4.2 → aimpg-0.4.3}/TODOS.md +0 -0
  15. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/__init__.py +0 -0
  16. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/attribution.py +0 -0
  17. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/coach.py +0 -0
  18. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/codex_logs.py +0 -0
  19. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/cost.py +0 -0
  20. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/cursor_usage.py +0 -0
  21. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/durable.py +0 -0
  22. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/energy.py +0 -0
  23. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/equivalence.py +0 -0
  24. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/equivalences.json +0 -0
  25. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/factors.json +0 -0
  26. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/gitkept.py +0 -0
  27. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/ledger.py +0 -0
  28. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/live.py +0 -0
  29. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/logs.py +0 -0
  30. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/model.py +0 -0
  31. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/prices.json +0 -0
  32. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/receipt.py +0 -0
  33. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/__init__.py +0 -0
  34. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/cli.py +0 -0
  35. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/fake_agent.py +0 -0
  36. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/hints.py +0 -0
  37. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/picker.py +0 -0
  38. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/proxy.py +0 -0
  39. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/run.py +0 -0
  40. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/sandbox.py +0 -0
  41. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/select.py +0 -0
  42. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/setups.py +0 -0
  43. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/stats.py +0 -0
  44. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/replay/workspace.py +0 -0
  45. {aimpg-0.4.2 → aimpg-0.4.3}/aimpg/share.py +0 -0
  46. {aimpg-0.4.2 → aimpg-0.4.3}/docs/STRATEGY.md +0 -0
  47. {aimpg-0.4.2 → aimpg-0.4.3}/docs/designs/aimpg-design.md +0 -0
  48. {aimpg-0.4.2 → aimpg-0.4.3}/docs/designs/scoreboard-brief.md +0 -0
  49. {aimpg-0.4.2 → aimpg-0.4.3}/evals/attribution_eval.py +0 -0
  50. {aimpg-0.4.2 → aimpg-0.4.3}/evals/label.py +0 -0
  51. {aimpg-0.4.2 → aimpg-0.4.3}/spikes/agent/RESULTS.md +0 -0
  52. {aimpg-0.4.2 → aimpg-0.4.3}/spikes/agent/allowlist_proxy.py +0 -0
  53. {aimpg-0.4.2 → aimpg-0.4.3}/spikes/agent/make_profile.py +0 -0
  54. {aimpg-0.4.2 → aimpg-0.4.3}/spikes/egress/RESULTS.md +0 -0
  55. {aimpg-0.4.2 → aimpg-0.4.3}/spikes/egress/allowlist_proxy.py +0 -0
  56. {aimpg-0.4.2 → aimpg-0.4.3}/spikes/egress/replay.sb +0 -0
  57. {aimpg-0.4.2 → aimpg-0.4.3}/tests/gitrepo.py +0 -0
  58. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_attribution.py +0 -0
  59. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_coach.py +0 -0
  60. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_codex_logs.py +0 -0
  61. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_cost_and_equivalence.py +0 -0
  62. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_cursor_usage.py +0 -0
  63. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_durable.py +0 -0
  64. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_energy.py +0 -0
  65. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_eval_scoring.py +0 -0
  66. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_gitkept.py +0 -0
  67. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_live.py +0 -0
  68. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_logs.py +0 -0
  69. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_perf.py +0 -0
  70. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_replay_cli.py +0 -0
  71. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_replay_e2e.py +0 -0
  72. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_replay_hints.py +0 -0
  73. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_replay_picker.py +0 -0
  74. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_replay_proxy.py +0 -0
  75. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_replay_sandbox.py +0 -0
  76. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_replay_stats.py +0 -0
  77. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_sandbox_required.py +0 -0
  78. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_share.py +0 -0
  79. {aimpg-0.4.2 → aimpg-0.4.3}/tests/test_verify.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: aimpg
3
- Version: 0.4.2
3
+ Version: 0.4.3
4
4
  Summary: Real-world energy per solved task for AI coding agents
5
5
  Project-URL: Homepage, https://github.com/kumarganduri/aimpg
6
6
  Project-URL: Issues, https://github.com/kumarganduri/aimpg/issues
@@ -115,6 +115,8 @@ That's the real result on the author's Legwork repo: no saving shown, far from t
115
115
 
116
116
  Every run writes a **record** (`~/.aimpg/verify/*.record.json`) holding versions, both setups, each run's tokens, $ and outcome, and the verdict, but never code, diffs or commit messages. Repo and commits are hashed unless you add `--public`. Anyone can recheck the math and the record's consistency for free with `aimpg verify --check record.json` (it can't prove the runs happened), and rerun a public record on their own machine with `aimpg verify --rerun record.json --repo <clone>`, which first shows every command, hook and prompt the record would run and asks before running them. Private records keep only fingerprints (SHA-256) of your prompts, CLAUDE.md, settings and commands, never the text. Records also list every redone attempt and excluded commit.
117
117
 
118
+ **Share it:** `aimpg submit <record>` publishes a stripped-down copy on the [public scoreboard](https://kumarganduri.github.io/aimpg-scoreboard) (no code, prompts, commit ids or repo name). It shows the exact file first and asks, then opens a pull request. Results only count once enough repos and people back them, and only independent reruns make the headline.
119
+
118
120
  Same model: the energy test decides (it must hold at every corner of the energy ranges). Different models or agents: cost per solved task decides, because model sizes are secret. A challenger that solves more than 10 points fewer tasks is never "supported". Codex runs are measured from its own logs and priced from OpenAI's published prices. Codex logs in with an OpenAI **API key** (a ChatGPT subscription login can't be used inside the sandbox); the key is written to the run's throwaway folder and deleted when the run ends. Other commands are judged on solve rate and time only.
119
121
 
120
122
  ### 5. Show the cost of each pull request
@@ -95,6 +95,8 @@ That's the real result on the author's Legwork repo: no saving shown, far from t
95
95
 
96
96
  Every run writes a **record** (`~/.aimpg/verify/*.record.json`) holding versions, both setups, each run's tokens, $ and outcome, and the verdict, but never code, diffs or commit messages. Repo and commits are hashed unless you add `--public`. Anyone can recheck the math and the record's consistency for free with `aimpg verify --check record.json` (it can't prove the runs happened), and rerun a public record on their own machine with `aimpg verify --rerun record.json --repo <clone>`, which first shows every command, hook and prompt the record would run and asks before running them. Private records keep only fingerprints (SHA-256) of your prompts, CLAUDE.md, settings and commands, never the text. Records also list every redone attempt and excluded commit.
97
97
 
98
+ **Share it:** `aimpg submit <record>` publishes a stripped-down copy on the [public scoreboard](https://kumarganduri.github.io/aimpg-scoreboard) (no code, prompts, commit ids or repo name). It shows the exact file first and asks, then opens a pull request. Results only count once enough repos and people back them, and only independent reruns make the headline.
99
+
98
100
  Same model: the energy test decides (it must hold at every corner of the energy ranges). Different models or agents: cost per solved task decides, because model sizes are secret. A challenger that solves more than 10 points fewer tasks is never "supported". Codex runs are measured from its own logs and priced from OpenAI's published prices. Codex logs in with an OpenAI **API key** (a ChatGPT subscription login can't be used inside the sandbox); the key is written to the run's throwaway folder and deleted when the run ends. Other commands are judged on solve rate and time only.
99
101
 
100
102
  ### 5. Show the cost of each pull request
@@ -20,6 +20,7 @@ from aimpg.gitkept import DAY
20
20
  from aimpg.logs import DEFAULT_ROOT, iter_log_files, parse_logs
21
21
  from aimpg.receipt import render
22
22
  from aimpg.replay import cli as replay_cli
23
+ from aimpg import scoreboard as scoreboard_mod
23
24
  from aimpg.replay import verify as verify_mod
24
25
 
25
26
 
@@ -73,11 +74,14 @@ def main(argv: list[str] | None = None) -> int:
73
74
 
74
75
  replay_cli.add_parser(sub)
75
76
  verify_mod.add_parser(sub)
77
+ scoreboard_mod.add_parsers(sub)
76
78
  args = parser.parse_args(argv)
77
79
  if args.command == "replay":
78
80
  return replay_cli.main(args)
79
81
  if args.command == "verify":
80
82
  return verify_mod.main(args)
83
+ if args.command in ("submit", "scoreboard"):
84
+ return scoreboard_mod.main(args)
81
85
  if args.command in ("statusline", "hook", "live"):
82
86
  from aimpg import live as live_mod
83
87
 
@@ -211,6 +211,12 @@ def _spec(setup: S.Setup, public: bool) -> dict:
211
211
  return spec
212
212
 
213
213
 
214
+ def _repo_id(repo: str) -> str | None:
215
+ from aimpg.scoreboard import repo_id
216
+
217
+ return repo_id(repo)
218
+
219
+
214
220
  def build_record(records: list[Record], *, claim: str, baseline: S.Setup, challenger: S.Setup, repo: str,
215
221
  task_mode: str, public: bool, extra: dict | None = None,
216
222
  attempts: list[Record] | None = None, excluded: set[str] | None = None) -> dict:
@@ -233,6 +239,7 @@ def build_record(records: list[Record], *, claim: str, baseline: S.Setup, challe
233
239
  "claim": claim,
234
240
  "public": public,
235
241
  "repo": _remote(repo) if public else hide(os.path.realpath(repo)),
242
+ "repo_id": _repo_id(repo), # keyed with a local secret: lets the scoreboard count repos
236
243
  "task_mode": task_mode,
237
244
  "versions": _versions(),
238
245
  "baseline": _spec(baseline, public),
@@ -518,9 +525,12 @@ def _run(args) -> int:
518
525
  if batch.stopped:
519
526
  print(f"STOPPED EARLY: an account can't make calls ({batch.stopped}). Fix it, then rerun with --resume {results}")
520
527
  records, excluded = runner.load_results(results)
528
+ # A rerun of a scoreboard upload names it by the upload's SHA-256 (its file name there)
529
+ rerun_of = hashlib.sha256(args.rerun.read_bytes()).hexdigest() if prior and prior.get("aimpg_upload") else None
521
530
  record = build_record(records, claim=claim, baseline=baseline, challenger=challenger, repo=str(repo),
522
531
  task_mode=task_mode, public=args.public if not prior else True,
523
- attempts=runner.all_attempts(results), excluded=excluded)
532
+ attempts=runner.all_attempts(results), excluded=excluded,
533
+ extra={"rerun_of": rerun_of} if rerun_of else None)
524
534
  out = results.with_suffix(".record.json")
525
535
  out.write_text(json.dumps(record, indent=1))
526
536
  print()
@@ -0,0 +1,621 @@
1
+ """The public scoreboard: upload view, validation, aggregation and the static page.
2
+
3
+ The scoreboard is a separate public repo (records + CI + GitHub Pages). All of
4
+ its rules live here so they are tested with aimpg and pinned by version.
5
+ Decisions: docs/designs/scoreboard-brief.md.
6
+
7
+ local record (~/.aimpg/verify/*.record.json, full)
8
+ │ upload_view(): drops texts, per-request tokens and commit ids; rounds the rest
9
+ ▼
10
+ upload (records/<github-login>/<sha256>.json in the scoreboard repo)
11
+ │ validate_upload(): CI on every pull request (data only, nothing executed)
12
+ ▼
13
+ build(): tiers, caps, thresholds → site/index.html + site/data.json
14
+
15
+ Privacy: a private upload holds model, setup, pass/fail, rounded cost, rounded
16
+ token sums, an energy range, coarse time and request count, and the week.
17
+ Never code, prompts, CLAUDE.md, commands, hosts, commit ids or the repo name.
18
+ A number is shown only when ≥5 repos and ≥3 submitters back it and no
19
+ submitter supplies more than half of its runs.
20
+
21
+ Trust: Reproduced (another account reran a public record and agreed),
22
+ Disputed (a rerun disagreed), Self-reported (everything else). Headlines use
23
+ reproduced records only; self-reported numbers are shown apart and labeled.
24
+ A vendor's records about its own product never count unless reproduced.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import hashlib
30
+ import hmac
31
+ import html
32
+ import json
33
+ import math
34
+ import os
35
+ import re
36
+ import secrets
37
+ import statistics
38
+ import subprocess
39
+ from collections import defaultdict
40
+ from datetime import datetime, timezone
41
+ from pathlib import Path
42
+
43
+ from aimpg.energy import ZERO, request_wh
44
+ from aimpg.model import Request, Usage
45
+
46
+ UPLOAD_VERSION = 2
47
+ SECRET = Path.home() / ".aimpg" / "id" # never leaves the machine; makes repo ids unguessable
48
+ MAX_BYTES = 1_000_000
49
+ MIN_REPOS, MIN_SUBMITTERS, MAX_SHARE = 5, 3, 0.5
50
+ REPOS_PER_ACCOUNT = 3
51
+ OUTCOMES = {"passed", "tests_failed", "timeout", "budget_hit", "agent_error"}
52
+ ANSWERS = {"SUPPORTED", "NOT SUPPORTED", "NOT PROVEN"}
53
+ # Setups everyone can name; anything else is published as "custom" or "cmd".
54
+ _CATALOG = re.compile(r"^(claude-code(\+rtk|\+terse)?(@[\w.\-]+)?|codex(@[\w.\-]+)?)$")
55
+ _CONTROL = re.compile(r"[\x00-\x08\x0b-\x1f\x7f‎‏‪-‮⁦-⁩]")
56
+ FOOTER = ("Self-reported numbers are not audited: `aimpg verify --check` confirms the arithmetic, "
57
+ "not that the runs happened. Only an independent rerun (Reproduced) is evidence.")
58
+
59
+
60
+ class UploadError(ValueError):
61
+ pass
62
+
63
+
64
+ # ---------------------------------------------------------------- identity
65
+
66
+ def _secret() -> bytes:
67
+ try:
68
+ return bytes.fromhex(SECRET.read_text().strip())
69
+ except (OSError, ValueError):
70
+ SECRET.parent.mkdir(parents=True, exist_ok=True)
71
+ value = secrets.token_bytes(32)
72
+ fd = os.open(SECRET, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
73
+ with os.fdopen(fd, "w") as fh:
74
+ fh.write(value.hex())
75
+ return value
76
+
77
+
78
+ def repo_id(repo: str) -> str | None:
79
+ """Same repo → same id on this machine, so the scoreboard counts repos, not records.
80
+
81
+ HMAC of the repo's first commit with a local secret: unlike a plain hash,
82
+ nobody can test it against public repos.
83
+ """
84
+ proc = subprocess.run(["git", "-C", repo, "rev-list", "--max-parents=0", "HEAD"], capture_output=True, text=True)
85
+ roots = sorted(proc.stdout.split())
86
+ if proc.returncode != 0 or not roots:
87
+ return None
88
+ return hmac.new(_secret(), roots[0].encode(), hashlib.sha256).hexdigest()[:24]
89
+
90
+
91
+ # ---------------------------------------------------------------- upload view
92
+
93
+ def _sig2(n: float) -> int:
94
+ if n <= 0:
95
+ return 0
96
+ digits = 2 - int(math.floor(math.log10(n))) - 1
97
+ return int(round(n, digits))
98
+
99
+
100
+ def _bucket(n: int) -> str:
101
+ return "1-5" if n <= 5 else "6-20" if n <= 20 else "21-50" if n <= 50 else "51+"
102
+
103
+
104
+ def _week(created: str) -> str:
105
+ year, week, _ = datetime.strptime(created, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=timezone.utc).isocalendar()
106
+ return f"{year}-W{week:02d}"
107
+
108
+
109
+ def _major_minor(version: str | None) -> str | None:
110
+ m = re.search(r"(\d+)\.(\d+)", version or "")
111
+ return f"{m.group(1)}.{m.group(2)}" if m else None
112
+
113
+
114
+ def claim_kind(baseline: dict, challenger: dict) -> str:
115
+ if baseline.get("agent", "claude") != challenger.get("agent", "claude"):
116
+ return "agent"
117
+ if (baseline.get("model") or None) != (challenger.get("model") or None):
118
+ return "model"
119
+ return "setup"
120
+
121
+
122
+ def _public_spec(spec: dict, public: bool) -> dict:
123
+ if public:
124
+ return spec
125
+ name = spec["name"] if _CATALOG.match(spec["name"]) else ("cmd" if spec.get("agent") == "cmd" else "custom")
126
+ out = {"name": name, "agent": spec.get("agent", "claude")}
127
+ if spec.get("model"):
128
+ out["model"] = spec["model"]
129
+ return out
130
+
131
+
132
+ def upload_view(record: dict, *, affiliation: str = "") -> dict:
133
+ """What leaves the machine. A --public record keeps what a rerun needs (its owner chose that)."""
134
+ if record.get("aimpg_record") != 1 or record.get("kind") != "verify":
135
+ raise UploadError("not an aimpg verify record")
136
+ public = bool(record.get("public"))
137
+ if not public and not record.get("repo_id"):
138
+ raise UploadError("this record has no repo id (made before aimpg 0.4.3); pass --repo to the repo it was made in")
139
+ commits = {}
140
+ runs = []
141
+ for r in record["runs"]:
142
+ usages = [Usage(*u) for u in r.get("usages", [])]
143
+ wh = ZERO
144
+ for u in usages:
145
+ wh = wh + request_wh(Request("x", "s", r["model"], 0.0, u, "/"))
146
+ sums = [sum(getattr(u, f) for u in usages) for f in ("fresh_in", "cache_write", "cache_read", "output")]
147
+ runs.append({
148
+ "commit": r["commit"] if public else commits.setdefault(r["commit"], len(commits)),
149
+ "setup": _public_spec({"name": r["setup"], "agent": r.get("agent", "claude")}, public)["name"],
150
+ "repeat": r["repeat"], "outcome": r["outcome"], "passed": r["passed"], "model": r["model"],
151
+ "agent": r.get("agent", "claude"), "cost_known": r.get("cost_known", True),
152
+ "cost_usd": round(r["cost_usd"], 2), "wall_s": int(round(r["wall_s"], -1)),
153
+ "tokens": [_sig2(x) for x in sums] if usages else None,
154
+ "wh": [round(wh.low, 2), round(wh.high, 2)] if usages else None,
155
+ "requests": _bucket(len(usages)) if usages else None,
156
+ "tokens_check": r.get("tokens_check", ""),
157
+ })
158
+ baseline, challenger = record["baseline"], record["challenger"]
159
+ v = record["verdict"]
160
+ out = {
161
+ "aimpg_upload": UPLOAD_VERSION,
162
+ "kind": "verify",
163
+ "week": _week(record["created"]),
164
+ "claim_kind": claim_kind(baseline, challenger),
165
+ "task_mode": record["task_mode"],
166
+ "public": public,
167
+ "repo_id": record.get("repo_id"),
168
+ "versions": {k: (val if k in ("aimpg", "factors", "prices") else _major_minor(val))
169
+ for k, val in record.get("versions", {}).items() if val},
170
+ "baseline": _public_spec(baseline, public),
171
+ "challenger": _public_spec(challenger, public),
172
+ "runs": runs,
173
+ "verdict": {"answer": v["answer"], "cost_ratio": v.get("cost_ratio"),
174
+ "energy": {k: v["energy"][k] for k in ("verdict", "effect_mid", "interval")} if v.get("energy") else None},
175
+ "superseded_attempts": len(record.get("superseded_attempts", [])),
176
+ "excluded_commits": len(record.get("excluded_commits", [])),
177
+ "rerun_of": record.get("rerun_of"),
178
+ "affiliation": affiliation.strip()[:100] or None,
179
+ }
180
+ if public:
181
+ out["repo"] = record.get("repo")
182
+ return out
183
+
184
+
185
+ def encode(upload: dict) -> bytes:
186
+ return (json.dumps(upload, indent=1, sort_keys=True) + "\n").encode()
187
+
188
+
189
+ def file_name(data: bytes) -> str:
190
+ return hashlib.sha256(data).hexdigest() + ".json"
191
+
192
+
193
+ # ---------------------------------------------------------------- validation (CI)
194
+
195
+ def _summary(runs: list[dict], setup: str) -> dict:
196
+ mine = [r for r in runs if r["setup"] == setup]
197
+ solved = sum(r["passed"] for r in mine)
198
+ priced = bool(mine) and all(r["cost_known"] for r in mine)
199
+ usd = sum(r["cost_usd"] for r in mine)
200
+ whs = [r["wh"] for r in mine if r.get("wh")]
201
+ return {"runs": len(mine), "solved": solved, "solve_rate": solved / len(mine) if mine else None,
202
+ "usd_per_solved": usd / solved if priced and solved else None,
203
+ "wh_per_solved": [sum(w[0] for w in whs) / solved, sum(w[1] for w in whs) / solved] if whs and solved else None}
204
+
205
+
206
+ def _strings(obj):
207
+ if isinstance(obj, str):
208
+ yield obj
209
+ elif isinstance(obj, dict):
210
+ for k, v in obj.items():
211
+ yield str(k)
212
+ yield from _strings(v)
213
+ elif isinstance(obj, list):
214
+ for v in obj:
215
+ yield from _strings(v)
216
+
217
+
218
+ def validate_upload(path: Path, data: bytes, *, author: str | None = None) -> list[str]:
219
+ """Problems with one submitted file (empty = accept). Pure data checks; nothing is executed."""
220
+ problems = []
221
+ if len(data) > MAX_BYTES:
222
+ return [f"{path.name}: larger than {MAX_BYTES} bytes"]
223
+ if path.name != file_name(data):
224
+ problems.append(f"{path.name}: the file name must be the SHA-256 of its content ({file_name(data)})")
225
+ if author is not None and path.parent.name != author:
226
+ problems.append(f"{path}: must be under records/{author}/ (the pull request's author)")
227
+ try:
228
+ u = json.loads(data)
229
+ except ValueError:
230
+ return problems + [f"{path.name}: not JSON"]
231
+ if not isinstance(u, dict) or u.get("aimpg_upload") != UPLOAD_VERSION or u.get("kind") != "verify":
232
+ return problems + [f"{path.name}: not an aimpg upload (version {UPLOAD_VERSION})"]
233
+ text = json.dumps(u)
234
+ if any(_CONTROL.search(t) for t in _strings(u)):
235
+ problems.append("control or direction-changing characters are not allowed")
236
+ if not u.get("public"):
237
+ if re.search(r"https?://|\b[0-9a-f]{40}\b", text):
238
+ problems.append("a private upload must not contain URLs or commit shas")
239
+ for side in ("baseline", "challenger"):
240
+ extra = set(u.get(side, {})) - {"name", "agent", "model"}
241
+ if extra:
242
+ problems.append(f"{side}: private uploads carry only name, agent and model (found {sorted(extra)})")
243
+ if not u.get("repo_id"):
244
+ problems.append("missing repo_id")
245
+ for side in ("baseline", "challenger"):
246
+ if not re.fullmatch(r"[\w.+@\-]{1,60}", str(u.get(side, {}).get("name", ""))):
247
+ problems.append(f"{side}: name must be letters, digits and .+@- only")
248
+ if not re.fullmatch(r"\d{4}-W\d{2}", str(u.get("week", ""))):
249
+ problems.append("week must look like 2026-W40")
250
+ if u.get("affiliation") is not None and (not isinstance(u["affiliation"], str) or len(u["affiliation"]) > 100):
251
+ problems.append("affiliation: text up to 100 characters")
252
+ runs = u.get("runs")
253
+ if not isinstance(runs, list) or not runs:
254
+ return problems + ["no runs"]
255
+ names = {u.get("baseline", {}).get("name"), u.get("challenger", {}).get("name")}
256
+ seen = set()
257
+ for i, r in enumerate(runs):
258
+ if r.get("outcome") not in OUTCOMES:
259
+ problems.append(f"run {i}: unknown outcome")
260
+ if bool(r.get("passed")) != (r.get("outcome") == "passed"):
261
+ problems.append(f"run {i}: passed doesn't match outcome")
262
+ if r.get("setup") not in names:
263
+ problems.append(f"run {i}: setup is neither baseline nor challenger")
264
+ if not isinstance(r.get("cost_usd"), (int, float)) or r["cost_usd"] < 0:
265
+ problems.append(f"run {i}: bad cost")
266
+ if r.get("tokens") is not None and not (len(r["tokens"]) == 4 and all(isinstance(x, int) and x >= 0 for x in r["tokens"])):
267
+ problems.append(f"run {i}: tokens must be four non-negative whole numbers")
268
+ key = (str(r.get("commit")), r.get("setup"), r.get("repeat"))
269
+ if key in seen:
270
+ problems.append(f"run {i}: duplicate")
271
+ seen.add(key)
272
+ commits, repeats = {k[0] for k in seen}, {k[2] for k in seen}
273
+ if any((c, s, n) not in seen for c in commits for s in names for n in repeats):
274
+ problems.append("incomplete: every commit × setup × repeat must be present (submit all runs, including failures)")
275
+ v = u.get("verdict") or {}
276
+ if v.get("answer") not in ANSWERS:
277
+ problems.append("verdict answer must be SUPPORTED, NOT SUPPORTED or NOT PROVEN")
278
+ if not problems: # re-check what can be re-checked from the upload itself
279
+ b, c = _summary(runs, u["baseline"]["name"]), _summary(runs, u["challenger"]["name"])
280
+ if v["answer"] == "SUPPORTED" and (not c["solved"] or b["solve_rate"] - c["solve_rate"] > 0.10):
281
+ problems.append("verdict says SUPPORTED but the challenger solves fewer tasks")
282
+ if v.get("cost_ratio") and b["usd_per_solved"] and c["usd_per_solved"]:
283
+ ratio = c["usd_per_solved"] / b["usd_per_solved"]
284
+ if abs(ratio - v["cost_ratio"]) > 0.05 * max(1.0, v["cost_ratio"]):
285
+ problems.append(f"cost ratio {v['cost_ratio']} doesn't match the runs ({ratio:.3f})")
286
+ return problems
287
+
288
+
289
+ # ---------------------------------------------------------------- aggregation
290
+
291
+ def load(records_dir: Path) -> list[dict]:
292
+ """[{submitter, sha, upload}] from records/<login>/<sha256>.json."""
293
+ out = []
294
+ for path in sorted(Path(records_dir).glob("*/*.json")):
295
+ out.append({"submitter": path.parent.name, "sha": path.stem, "upload": json.loads(path.read_text())})
296
+ return out
297
+
298
+
299
+ def _claim(u: dict) -> tuple[str, str]:
300
+ return u["baseline"]["name"], u["challenger"]["name"]
301
+
302
+
303
+ def tiers(entries: list[dict]) -> dict[str, str]:
304
+ """sha → reproduced | disputed | self-reported | rerun."""
305
+ by_sha = {e["sha"]: e for e in entries}
306
+ reruns = defaultdict(list)
307
+ out = {}
308
+ for e in entries:
309
+ target = e["upload"].get("rerun_of")
310
+ if target:
311
+ out[e["sha"]] = "rerun"
312
+ if target in by_sha and by_sha[target]["submitter"] != e["submitter"]:
313
+ reruns[target].append(e["upload"]["verdict"]["answer"])
314
+ for e in entries:
315
+ if e["sha"] in out:
316
+ continue
317
+ answers = reruns.get(e["sha"], [])
318
+ mine = e["upload"]["verdict"]["answer"]
319
+ out[e["sha"]] = "disputed" if any(a != mine for a in answers) else "reproduced" if answers else "self-reported"
320
+ return out
321
+
322
+
323
+ def _metrics(u: dict) -> dict:
324
+ b, c = _summary(u["runs"], u["baseline"]["name"]), _summary(u["runs"], u["challenger"]["name"])
325
+ return {"answer": u["verdict"]["answer"],
326
+ "cost_ratio": c["usd_per_solved"] / b["usd_per_solved"] if b["usd_per_solved"] and c["usd_per_solved"] else None,
327
+ "solve_diff": (c["solve_rate"] or 0) - (b["solve_rate"] or 0),
328
+ "runs": len(u["runs"])}
329
+
330
+
331
+ def aggregate(entries: list[dict], vendors: dict[str, list[str]] | None = None) -> dict:
332
+ vendors = vendors or {}
333
+ tier = tiers(entries)
334
+ groups = {"reproduced": defaultdict(list), "self-reported": defaultdict(list)}
335
+ collecting = set()
336
+ for e in entries:
337
+ u, t = e["upload"], tier[e["sha"]]
338
+ claim = _claim(u)
339
+ if t in ("rerun", "disputed") or "custom" in claim or "cmd" in claim:
340
+ continue
341
+ if e["submitter"] in vendors.get(claim[1], []) and t != "reproduced":
342
+ continue # a vendor's own claim counts only once someone else reproduces it
343
+ groups[t][claim].append(e)
344
+ cells = {}
345
+ for t, by_claim in groups.items():
346
+ rows = []
347
+ for claim, es in sorted(by_claim.items()):
348
+ # one unit per repo (newest record), at most REPOS_PER_ACCOUNT repos per account
349
+ per_repo = {}
350
+ for e in sorted(es, key=lambda e: e["upload"]["week"]):
351
+ per_repo[e["upload"].get("repo_id") or e["upload"].get("repo")] = e
352
+ kept, per_account = [], defaultdict(int)
353
+ for e in sorted(per_repo.values(), key=lambda e: e["upload"]["week"], reverse=True):
354
+ if per_account[e["submitter"]] < REPOS_PER_ACCOUNT:
355
+ per_account[e["submitter"]] += 1
356
+ kept.append(e)
357
+ metrics = [(e["submitter"], _metrics(e["upload"])) for e in kept]
358
+ runs = defaultdict(int)
359
+ for s, m in metrics:
360
+ runs[s] += m["runs"]
361
+ total = sum(runs.values())
362
+ enough = (len(kept) >= MIN_REPOS and len(runs) >= MIN_SUBMITTERS
363
+ and max(runs.values()) <= MAX_SHARE * total)
364
+ if not enough:
365
+ collecting.add(claim)
366
+ continue
367
+ ratios = [m["cost_ratio"] for _, m in metrics if m["cost_ratio"] is not None]
368
+ rows.append({
369
+ "baseline": claim[0], "challenger": claim[1],
370
+ "repos": len(kept), "submitters": len(runs),
371
+ "answers": {a: sum(m["answer"] == a for _, m in metrics) for a in sorted(ANSWERS)},
372
+ "median_cost_ratio": statistics.median(ratios) if ratios else None,
373
+ "median_solve_diff": statistics.median(m["solve_diff"] for _, m in metrics),
374
+ })
375
+ cells[t] = rows
376
+ shown = {(r["baseline"], r["challenger"]) for rows in cells.values() for r in rows}
377
+ public = [{"sha": e["sha"], "submitter": e["submitter"], "tier": tier[e["sha"]], "week": e["upload"]["week"],
378
+ "baseline": e["upload"]["baseline"]["name"], "challenger": e["upload"]["challenger"]["name"],
379
+ "answer": e["upload"]["verdict"]["answer"], "repo": e["upload"].get("repo")}
380
+ for e in entries if e["upload"].get("public")]
381
+ disputed = [{"sha": e["sha"], "submitter": e["submitter"], "challenger": e["upload"]["challenger"]["name"],
382
+ "answer": e["upload"]["verdict"]["answer"]} for e in entries if tier[e["sha"]] == "disputed"]
383
+ return {
384
+ "rules": {"min_repos": MIN_REPOS, "min_submitters": MIN_SUBMITTERS, "max_share": MAX_SHARE,
385
+ "repos_per_account": REPOS_PER_ACCOUNT},
386
+ "records": len(entries),
387
+ "reproduced": cells.get("reproduced", []),
388
+ "self_reported": cells.get("self-reported", []),
389
+ "collecting": sorted(f"{c} vs {b}" for b, c in collecting - shown),
390
+ # trusted first; reruns (evidence for another record) last
391
+ "public_records": sorted(sorted(public, key=lambda p: p["week"], reverse=True),
392
+ key=lambda p: ["reproduced", "disputed", "self-reported", "rerun"].index(p["tier"])),
393
+ "disputed": disputed,
394
+ "footer": FOOTER,
395
+ }
396
+
397
+
398
+ # ---------------------------------------------------------------- page
399
+
400
+ def _pct(x: float | None, signed: bool = True) -> str:
401
+ return "—" if x is None else (f"{x:+.0%}" if signed else f"{x:.0%}")
402
+
403
+
404
+ def _table(rows: list[dict]) -> str:
405
+ if not rows:
406
+ return '<p class="empty">Nothing here yet: no claim has enough repos and people behind it.</p>'
407
+ out = ['<table><thead><tr><th>Claim</th><th>Result across repos</th><th>Cost per solved task</th>'
408
+ '<th>Tasks solved</th><th>Based on</th></tr></thead><tbody>']
409
+ for r in rows:
410
+ a = r["answers"]
411
+ result = (f'<span class="yes">{a["SUPPORTED"]} supported</span> · <span class="no">{a["NOT SUPPORTED"]} not</span>'
412
+ f' · <span class="maybe">{a["NOT PROVEN"]} not proven</span>')
413
+ cost = "—" if r["median_cost_ratio"] is None else _pct(r["median_cost_ratio"] - 1)
414
+ out.append(f'<tr><td><b>{html.escape(r["challenger"])}</b><br><small>vs {html.escape(r["baseline"])}</small></td>'
415
+ f'<td>{result}</td><td>{cost} <small>median</small></td><td>{r["median_solve_diff"] * 100:+.0f} pts <small>median</small></td>'
416
+ f'<td>{r["repos"]} repos · {r["submitters"]} people</td></tr>')
417
+ return "\n".join(out + ["</tbody></table>"])
418
+
419
+
420
+ def render(data: dict, *, demo: bool = False) -> str:
421
+ esc = html.escape
422
+ public_rows = "".join(
423
+ f'<tr><td><span class="tier {esc(p["tier"])}">{esc(p["tier"])}</span></td><td>{esc(p["challenger"])} '
424
+ f'<small>vs {esc(p["baseline"])}</small></td><td>{esc(p["answer"])}</td><td>{esc(p["submitter"])}</td>'
425
+ f'<td>{esc(p["week"])}</td><td><a href="records/{esc(p["submitter"])}/{esc(p["sha"])}.json">record</a></td></tr>'
426
+ for p in data["public_records"]) or '<tr><td colspan="6" class="empty">No public records yet.</td></tr>'
427
+ collecting = ", ".join(esc(c) for c in data["collecting"]) or "—"
428
+ disputed = "".join(f"<li>{esc(d['challenger'])}: {esc(d['answer'])} by {esc(d['submitter'])}, a rerun disagreed "
429
+ f'(<a href="records/{esc(d["submitter"])}/{esc(d["sha"])}.json">record</a>)</li>'
430
+ for d in data["disputed"]) or "<li>None.</li>"
431
+ rules = data["rules"]
432
+ footer = re.sub(r"`([^`]+)`", r"<code>\1</code>", esc(data["footer"]))
433
+ banner = '<p class="demo">DEMO DATA: made-up records to preview the page. Not real results.</p>' if demo else ""
434
+ return f"""<!doctype html>
435
+ <html lang="en"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1">
436
+ <title>aimpg scoreboard</title>
437
+ <style>
438
+ :root{{--bg:#fbfaf7;--fg:#1d1d1b;--muted:#6b6a65;--line:#e4e1d8;--yes:#1f7a4d;--no:#b3261e;--maybe:#8a6d00;--chip:#efece4}}
439
+ @media (prefers-color-scheme:dark){{:root{{--bg:#151513;--fg:#ecebe6;--muted:#a3a19a;--line:#33322e;--yes:#5cc18f;--no:#f08a80;--maybe:#e0c063;--chip:#26251f}}}}
440
+ body{{background:var(--bg);color:var(--fg);font:16px/1.55 system-ui,-apple-system,sans-serif;margin:0}}
441
+ main{{max-width:960px;margin:0 auto;padding:32px 16px 64px}}
442
+ h1{{font-size:28px;margin:0 0 4px}} h2{{font-size:19px;margin:36px 0 8px}} p{{margin:6px 0}}
443
+ .lede{{color:var(--muted);max-width:680px}} small{{color:var(--muted)}}
444
+ .wrap{{overflow-x:auto}} table{{border-collapse:collapse;width:100%;font-size:15px}}
445
+ th,td{{text-align:left;padding:9px 10px;border-bottom:1px solid var(--line);vertical-align:top}}
446
+ th{{font-size:13px;color:var(--muted);font-weight:600}}
447
+ .yes{{color:var(--yes)}} .no{{color:var(--no)}} .maybe{{color:var(--maybe)}}
448
+ .tier{{font-size:12px;padding:2px 8px;border-radius:99px;background:var(--chip)}}
449
+ .tier.reproduced{{color:var(--yes)}} .tier.disputed{{color:var(--no)}}
450
+ .empty{{color:var(--muted)}} .demo{{background:var(--chip);padding:8px 12px;border-radius:8px;font-weight:600}}
451
+ footer{{margin-top:40px;color:var(--muted);font-size:14px;border-top:1px solid var(--line);padding-top:12px}}
452
+ code{{background:var(--chip);padding:1px 5px;border-radius:4px;font-size:14px}} a{{color:inherit}}
453
+ </style></head><body><main>
454
+ {banner}
455
+ <h1>aimpg scoreboard</h1>
456
+ <p class="lede">Do token-savers, prompts, models and agents really save money on real code? Each result reruns a
457
+ developer's own past commits both ways, and their own tests judge them. Made with
458
+ <a href="https://github.com/kumarganduri/aimpg">aimpg verify</a>; {data["records"]} records so far.</p>
459
+
460
+ <h2>Reproduced</h2>
461
+ <p class="lede">Counted only when someone else reran a public record and got the same answer.</p>
462
+ <div class="wrap">{_table(data["reproduced"])}</div>
463
+
464
+ <h2>Self-reported <small>(not audited)</small></h2>
465
+ <div class="wrap">{_table(data["self_reported"])}</div>
466
+
467
+ <h2>Collecting data</h2>
468
+ <p>{collecting}</p>
469
+ <p><small>A number appears once at least {rules["min_repos"]} repos and {rules["min_submitters"]} people back it, with no one
470
+ supplying more than {rules["max_share"]:.0%} of the runs. Each account counts for at most {rules["repos_per_account"]} repos per claim.
471
+ Results use the median repo, never an average.</small></p>
472
+
473
+ <h2>Disputed</h2><ul>{disputed}</ul>
474
+
475
+ <h2>Public records <small>(rerunnable)</small></h2>
476
+ <div class="wrap"><table><thead><tr><th>Tier</th><th>Claim</th><th>Answer</th><th>By</th><th>Week</th><th></th></tr></thead>
477
+ <tbody>{public_rows}</tbody></table></div>
478
+
479
+ <footer><p>{footer}</p>
480
+ <p>Add yours: <code>aimpg verify ...</code> then <code>aimpg submit</code>. Private uploads carry no code, prompts,
481
+ commit ids or repo names. Withdraw anytime by pull request.</p></footer>
482
+ </main></body></html>
483
+ """
484
+
485
+
486
+ def build(records_dir: Path, out_dir: Path, vendors_file: Path | None = None, *, demo: bool = False) -> dict:
487
+ vendors = json.loads(vendors_file.read_text()) if vendors_file and vendors_file.exists() else {}
488
+ data = aggregate(load(records_dir), vendors)
489
+ out_dir.mkdir(parents=True, exist_ok=True)
490
+ (out_dir / "data.json").write_text(json.dumps(data, indent=1))
491
+ (out_dir / "index.html").write_text(render(data, demo=demo))
492
+ return data
493
+
494
+
495
+ # ---------------------------------------------------------------- command line
496
+
497
+ SCOREBOARD_REPO = "kumarganduri/aimpg-scoreboard"
498
+
499
+
500
+ def add_parsers(sub) -> None:
501
+ s = sub.add_parser("submit", help="share a verify record on the public scoreboard (shows everything first, asks)")
502
+ s.add_argument("record", type=Path)
503
+ s.add_argument("--affiliation", default="", help="who you work for, if it relates to the claim (shown publicly)")
504
+ s.add_argument("--repo", type=Path, help="the repo the record was made in (only for records from before 0.4.3)")
505
+ s.add_argument("--to-dir", type=Path, help="write into a local scoreboard checkout instead of opening a pull request")
506
+ s.add_argument("--login", help="your GitHub username (with --to-dir; otherwise taken from gh)")
507
+ s.add_argument("--scoreboard", default=SCOREBOARD_REPO, help=f"scoreboard repo (default {SCOREBOARD_REPO})")
508
+ s.add_argument("--yes", action="store_true", help="skip the confirmation")
509
+
510
+ b = sub.add_parser("scoreboard", help="scoreboard maintenance (used by its CI)")
511
+ bsub = b.add_subparsers(dest="scoreboard_command", required=True)
512
+ v = bsub.add_parser("validate", help="check submitted files")
513
+ v.add_argument("files", type=Path, nargs="+")
514
+ v.add_argument("--author", help="pull request author: files must be under records/<author>/")
515
+ g = bsub.add_parser("build", help="aggregate records into the static page")
516
+ g.add_argument("records", type=Path)
517
+ g.add_argument("out", type=Path)
518
+ g.add_argument("--vendors", type=Path, help="vendors.json: {challenger: [github logins]}")
519
+ g.add_argument("--demo", action="store_true", help="label the page as demo data")
520
+
521
+
522
+ def main(args) -> int:
523
+ if args.command == "submit":
524
+ try:
525
+ return _submit(args)
526
+ except UploadError as exc:
527
+ print(f"submit: {exc}")
528
+ return 1
529
+ if args.scoreboard_command == "validate":
530
+ bad = 0
531
+ for f in args.files:
532
+ problems = validate_upload(f, f.read_bytes(), author=args.author)
533
+ print(f"{'✗' if problems else '✓'} {f}")
534
+ for p in problems:
535
+ print(f" {p}")
536
+ bad += bool(problems)
537
+ return 1 if bad else 0
538
+ data = build(args.records, args.out, args.vendors, demo=args.demo)
539
+ print(f"Built {args.out / 'index.html'} from {data['records']} records.")
540
+ return 0
541
+
542
+
543
+ def _submit(args) -> int:
544
+ from aimpg.replay.verify import check, integrity
545
+
546
+ record = json.loads(args.record.read_text())
547
+ if not record.get("repo_id") and args.repo:
548
+ record["repo_id"] = repo_id(str(args.repo))
549
+ ok, _ = check(record)
550
+ problems, _ = integrity(record)
551
+ if not ok or problems:
552
+ raise UploadError("the record fails `aimpg verify --check`; only unmodified records can be submitted")
553
+ upload = upload_view(record, affiliation=args.affiliation)
554
+ data = encode(upload)
555
+ name = file_name(data)
556
+ issues = validate_upload(Path("records") / "x" / name, data)
557
+ if issues:
558
+ raise UploadError("; ".join(issues))
559
+
560
+ print("This exact file will be published on the public scoreboard:\n")
561
+ print(data.decode())
562
+ if upload["public"]:
563
+ print("It's a --public record: it includes the repo URL, commit ids and setup texts, so anyone can rerun it.")
564
+ else:
565
+ print("Not included: your code, prompts, CLAUDE.md, commands, commit ids, repo name, exact token counts or dates.")
566
+ print("Your GitHub username will be shown with it. You can withdraw it anytime with a pull request that deletes it.")
567
+ if not args.yes and input("Publish it? [y/N] ").strip().lower() != "y":
568
+ print("Stopped. Nothing was sent.")
569
+ return 1
570
+
571
+ if args.to_dir:
572
+ if not args.login:
573
+ raise UploadError("--to-dir needs --login (your GitHub username)")
574
+ dest = args.to_dir / "records" / args.login / name
575
+ dest.parent.mkdir(parents=True, exist_ok=True)
576
+ dest.write_bytes(data)
577
+ print(f"Written to {dest}")
578
+ return 0
579
+ url = _open_pull_request(args.scoreboard, name, data, upload)
580
+ print(f"Pull request: {url}\nThe scoreboard's checks run on it; once merged, the page updates.")
581
+ return 0
582
+
583
+
584
+ def _gh(*argv: str, input_text: str | None = None) -> str:
585
+ proc = subprocess.run(["gh", *argv], capture_output=True, text=True, input=input_text)
586
+ if proc.returncode != 0:
587
+ raise UploadError(f"gh {argv[0]} failed: {(proc.stderr or proc.stdout).strip()[-300:]}")
588
+ return proc.stdout.strip()
589
+
590
+
591
+ def _open_pull_request(upstream: str, name: str, data: bytes, upload: dict) -> str:
592
+ """Fork (if needed), add the file on a new branch through the API (no clone), open the PR."""
593
+ import base64
594
+ import time
595
+
596
+ try:
597
+ login = _gh("api", "user", "-q", ".login")
598
+ except (UploadError, OSError):
599
+ raise UploadError("needs the GitHub CLI, logged in: brew install gh && gh auth login") from None
600
+ owner_repo = upstream
601
+ if login.lower() != upstream.split("/")[0].lower():
602
+ _gh("repo", "fork", upstream, "--clone=false")
603
+ owner_repo = f"{login}/{upstream.split('/')[1]}"
604
+ for _ in range(10): # forks are created asynchronously
605
+ try:
606
+ _gh("repo", "sync", owner_repo)
607
+ break
608
+ except UploadError:
609
+ time.sleep(3)
610
+ base = _gh("api", f"repos/{upstream}", "-q", ".default_branch")
611
+ sha = _gh("api", f"repos/{owner_repo}/git/ref/heads/{base}", "-q", ".object.sha")
612
+ branch = f"aimpg-{name[:12]}"
613
+ _gh("api", "-X", "POST", f"repos/{owner_repo}/git/refs", "-f", f"ref=refs/heads/{branch}", "-f", f"sha={sha}")
614
+ _gh("api", "-X", "PUT", f"repos/{owner_repo}/contents/records/{login}/{name}",
615
+ "-f", f"message=verify: {upload['challenger']['name']} vs {upload['baseline']['name']}",
616
+ "-f", f"content={base64.b64encode(data).decode()}", "-f", f"branch={branch}")
617
+ body = (f"**Claim:** {upload['challenger']['name']} vs {upload['baseline']['name']} → "
618
+ f"{upload['verdict']['answer']}\n\n**Affiliation:** {upload.get('affiliation') or 'none stated'}\n\n"
619
+ f"Submitted with `aimpg submit`. {FOOTER}")
620
+ return _gh("pr", "create", "--repo", upstream, "--head", f"{login}:{branch}" if owner_repo != upstream else branch,
621
+ "--base", base, "--title", f"verify: {upload['challenger']['name']} vs {upload['baseline']['name']}", "--body", body)
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "aimpg"
3
- version = "0.4.2"
3
+ version = "0.4.3"
4
4
  description = "Real-world energy per solved task for AI coding agents"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -94,3 +94,6 @@ def _never_read_the_real_agent_logs(tmp_path_factory, monkeypatch):
94
94
  monkeypatch.setattr(ledger, "LEDGER", tmp_path_factory.mktemp("ledger") / "ledger.json")
95
95
  monkeypatch.setattr(replay_cli, "STATE", tmp_path_factory.mktemp("replay-state"))
96
96
  monkeypatch.setattr(verify, "STATE", tmp_path_factory.mktemp("verify-state"))
97
+ from aimpg import scoreboard
98
+
99
+ monkeypatch.setattr(scoreboard, "SECRET", tmp_path_factory.mktemp("aimpg-id") / "id")
@@ -0,0 +1,201 @@
1
+ import json
2
+ import subprocess
3
+
4
+ import pytest
5
+
6
+ from aimpg import scoreboard as sb
7
+ from aimpg.cli import main
8
+ from aimpg.replay import setups as S
9
+ from aimpg.replay.run import Record
10
+ from aimpg.replay.verify import build_record
11
+
12
+ SONNET = "claude-sonnet-5-5"
13
+
14
+
15
+ def run(commit, setup, *, passed=True, usages=None, cost=0.10, model=SONNET, repeat=0):
16
+ usages = usages or [[5, 3000, 9000, 400]] * 3
17
+ return Record(f"{commit}-{setup}-{repeat}", "/repo", commit, setup, repeat, "passed" if passed else "tests_failed",
18
+ passed, model, cost, 33.0, usages, "ok", "", True, "claude")
19
+
20
+
21
+ def record(tmp_path, *, challenger=S.TERSE, chal_cost=0.10, chal_pass=True, public=False, n=3):
22
+ repo = tmp_path / f"repo{len(list(tmp_path.iterdir()))}"
23
+ repo.mkdir()
24
+ subprocess.run(["git", "init", "-q", str(repo)], check=True)
25
+ subprocess.run(["git", "-C", str(repo), "-c", "user.email=a@b", "-c", "user.name=a", "commit", "-q", "--allow-empty", "-m", f"root {repo.name}"], check=True)
26
+ runs = [run(f"{i:040x}", "claude-code") for i in range(n)] + [
27
+ run(f"{i:040x}", challenger.name, cost=chal_cost, passed=chal_pass) for i in range(n)]
28
+ return build_record(runs, claim="", baseline=S.BASELINE, challenger=challenger, repo=str(repo), task_mode="tests", public=public)
29
+
30
+
31
+ def put(records_dir, login, upload):
32
+ data = sb.encode(upload)
33
+ path = records_dir / login / sb.file_name(data)
34
+ path.parent.mkdir(parents=True, exist_ok=True)
35
+ path.write_bytes(data)
36
+ return path.stem
37
+
38
+
39
+ # ---------------------------------------------------------------- upload view
40
+
41
+ def test_private_upload_carries_no_texts_ids_or_exact_numbers(tmp_path):
42
+ mine = S.custom(append_prompt="internal: ask #team-secrets", claude_md="db.internal.corp")
43
+ rec = record(tmp_path, challenger=mine)
44
+ up = sb.upload_view(rec)
45
+ text = json.dumps(up)
46
+ assert "internal" not in text and "sha256" not in text and f"{0:040x}" not in text
47
+ assert up["challenger"] == {"name": "custom", "agent": "claude"}
48
+ assert {r["commit"] for r in up["runs"]} == {0, 1, 2} # run-local indexes, not commit ids
49
+ assert up["runs"][0]["tokens"] == [15, 9000, 27000, 1200] # sums, 2 significant figures
50
+ assert up["runs"][0]["requests"] == "1-5" and up["runs"][0]["wall_s"] == 30
51
+ assert up["week"].startswith("20") and "-W" in up["week"] and "created" not in up
52
+ assert up["repo_id"] and len(up["repo_id"]) == 24
53
+ assert sb.validate_upload(tmp_path / "x" / sb.file_name(sb.encode(up)), sb.encode(up)) == []
54
+
55
+
56
+ def test_repo_id_is_stable_per_repo_and_keyed_by_a_local_secret(tmp_path):
57
+ a = record(tmp_path)
58
+ repo = tmp_path / "repo0"
59
+ assert sb.repo_id(str(repo)) == a["repo_id"]
60
+ sb.SECRET.write_text("00" * 32) # another machine's secret → a different id for the same repo
61
+ assert sb.repo_id(str(repo)) != a["repo_id"]
62
+
63
+
64
+ def test_catalog_setups_keep_their_names():
65
+ for name in ("claude-code+rtk", "claude-code@haiku", "codex@gpt-6.1-sol"):
66
+ assert sb._public_spec({"name": name, "agent": "claude"}, False)["name"] == name
67
+
68
+
69
+ # ---------------------------------------------------------------- validation
70
+
71
+ def valid(tmp_path, **kw):
72
+ up = sb.upload_view(record(tmp_path, **kw))
73
+ return up
74
+
75
+
76
+ def problems(tmp_path, up, author=None, folder="me"):
77
+ data = sb.encode(up)
78
+ return sb.validate_upload(tmp_path / folder / sb.file_name(data), data, author=author)
79
+
80
+
81
+ def test_validation_rejects_tampering_and_leaks(tmp_path):
82
+ up = valid(tmp_path)
83
+ assert problems(tmp_path, up, author="me") == []
84
+ assert problems(tmp_path, up, author="someone-else") # wrong folder for the PR author
85
+ data = sb.encode(up)
86
+ assert sb.validate_upload(tmp_path / "me" / "wrong.json", data) # name must be the content hash
87
+
88
+ bad = json.loads(json.dumps(up))
89
+ bad["runs"][0]["passed"] = not bad["runs"][0]["passed"]
90
+ assert any("passed" in p for p in problems(tmp_path, bad))
91
+
92
+ leak = json.loads(json.dumps(up))
93
+ leak["challenger"]["claude_md"] = "secret"
94
+ leak["affiliation"] = "see https://corp.example"
95
+ found = problems(tmp_path, leak)
96
+ assert any("only name, agent and model" in p for p in found) and any("URLs" in p for p in found)
97
+
98
+ partial = json.loads(json.dumps(up))
99
+ partial["runs"] = partial["runs"][:-1]
100
+ assert any("incomplete" in p for p in problems(tmp_path, partial))
101
+
102
+ bidi = json.loads(json.dumps(up))
103
+ bidi["affiliation"] = "acme‮"
104
+ assert any("control" in p for p in problems(tmp_path, bidi))
105
+
106
+ html_name = json.loads(json.dumps(up))
107
+ html_name["challenger"]["name"] = "<script>"
108
+ assert any("letters, digits" in p for p in problems(tmp_path, html_name))
109
+
110
+
111
+ def test_validation_rechecks_the_verdict_against_the_runs(tmp_path):
112
+ up = valid(tmp_path, chal_pass=False)
113
+ up["verdict"]["answer"] = "SUPPORTED"
114
+ assert any("solves fewer" in p for p in problems(tmp_path, up))
115
+
116
+
117
+ # ---------------------------------------------------------------- aggregation
118
+
119
+ def many(tmp_path, records_dir, logins, **kw):
120
+ shas = []
121
+ for login in logins:
122
+ shas.append(put(records_dir, login, sb.upload_view(record(tmp_path, **kw))))
123
+ return shas
124
+
125
+
126
+ def test_numbers_appear_only_with_5_repos_and_3_people(tmp_path):
127
+ d = tmp_path / "records"
128
+ many(tmp_path, d, ["ann", "bob", "cat", "dan"], chal_cost=0.05)
129
+ data = sb.aggregate(sb.load(d))
130
+ assert data["self_reported"] == [] and data["collecting"] == ["claude-code+terse vs claude-code"]
131
+ many(tmp_path, d, ["eve"], chal_cost=0.05)
132
+ (row,) = sb.aggregate(sb.load(d))["self_reported"]
133
+ assert row["repos"] == 5 and row["submitters"] == 5
134
+ assert row["answers"]["NOT PROVEN"] == 5 and row["median_cost_ratio"] == pytest.approx(0.5) # same model: energy decides
135
+ assert sb.aggregate(sb.load(d))["reproduced"] == [] # self-reported never reaches the headline
136
+
137
+
138
+ def test_one_account_cannot_carry_a_claim(tmp_path):
139
+ d = tmp_path / "records"
140
+ many(tmp_path, d, ["ann"] * 6 + ["bob", "cat"]) # ann's 6 repos are capped at 3, and 3 of 5 runs > 50%
141
+ data = sb.aggregate(sb.load(d))
142
+ assert data["self_reported"] == []
143
+
144
+
145
+ def test_vendor_claims_dont_count_unless_reproduced(tmp_path):
146
+ d = tmp_path / "records"
147
+ many(tmp_path, d, ["rtk-inc", "ann", "bob", "cat", "dan"], challenger=S.RTK)
148
+ assert sb.aggregate(sb.load(d), {"claude-code+rtk": ["rtk-inc"]})["self_reported"] == []
149
+ assert sb.aggregate(sb.load(d))["self_reported"] # without the vendor list it would count
150
+
151
+
152
+ def test_tiers_from_independent_reruns(tmp_path):
153
+ d = tmp_path / "records"
154
+ original = put(d, "ann", sb.upload_view(record(tmp_path, public=True)))
155
+ same_person = sb.upload_view(record(tmp_path, public=True))
156
+ same_person["rerun_of"] = original
157
+ put(d, "ann", same_person)
158
+ assert sb.tiers(sb.load(d))[original] == "self-reported" # rerunning your own record proves nothing
159
+ agree = sb.upload_view(record(tmp_path, public=True))
160
+ agree["rerun_of"] = original
161
+ put(d, "bob", agree)
162
+ assert sb.tiers(sb.load(d))[original] == "reproduced"
163
+ disagree = sb.upload_view(record(tmp_path, public=True, chal_pass=False))
164
+ disagree["rerun_of"] = original
165
+ put(d, "cat", disagree)
166
+ assert sb.tiers(sb.load(d))[original] == "disputed"
167
+
168
+
169
+ def test_page_escapes_and_labels(tmp_path):
170
+ d = tmp_path / "records"
171
+ up = sb.upload_view(record(tmp_path, public=True))
172
+ up["affiliation"] = "<b>acme</b>"
173
+ put(d, "ann", up)
174
+ data = sb.build(d, tmp_path / "site", demo=True)
175
+ page = (tmp_path / "site" / "index.html").read_text()
176
+ assert "DEMO DATA" in page and "not audited" in page and data["records"] == 1
177
+ assert "<b>acme" not in page
178
+
179
+
180
+ # ---------------------------------------------------------------- submit
181
+
182
+ def test_submit_to_a_local_scoreboard_checkout(tmp_path, monkeypatch, capsys):
183
+ rec = record(tmp_path, challenger=S.custom(append_prompt="secret prompt"))
184
+ path = tmp_path / "r.record.json"
185
+ path.write_text(json.dumps(rec))
186
+ monkeypatch.setattr("builtins.input", lambda prompt="": "y")
187
+ board = tmp_path / "board"
188
+ assert main(["submit", str(path), "--to-dir", str(board), "--login", "ann"]) == 0
189
+ out = capsys.readouterr().out
190
+ assert "This exact file will be published" in out and "secret prompt" not in out
191
+ (written,) = list((board / "records" / "ann").glob("*.json"))
192
+ assert main(["scoreboard", "validate", str(written), "--author", "ann"]) == 0
193
+
194
+
195
+ def test_submit_refuses_an_edited_record(tmp_path, monkeypatch, capsys):
196
+ rec = record(tmp_path)
197
+ rec["runs"][0]["passed"] = not rec["runs"][0]["passed"]
198
+ path = tmp_path / "r.record.json"
199
+ path.write_text(json.dumps(rec))
200
+ assert main(["submit", str(path), "--to-dir", str(tmp_path), "--login", "ann", "--yes"]) == 1
201
+ assert "fails `aimpg verify --check`" in capsys.readouterr().out
@@ -4,7 +4,7 @@ requires-python = ">=3.10"
4
4
 
5
5
  [[package]]
6
6
  name = "aimpg"
7
- version = "0.4.2"
7
+ version = "0.4.3"
8
8
  source = { editable = "." }
9
9
 
10
10
  [package.dev-dependencies]
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes