aimpg 0.4.0__tar.gz → 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. aimpg-0.4.2/.github/workflows/test.yml +34 -0
  2. {aimpg-0.4.0 → aimpg-0.4.2}/PKG-INFO +4 -4
  3. {aimpg-0.4.0 → aimpg-0.4.2}/README.md +3 -3
  4. aimpg-0.4.2/TODOS.md +9 -0
  5. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/cli.py +2 -1
  6. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/fake_agent.py +11 -6
  7. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/proxy.py +18 -2
  8. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/run.py +12 -0
  9. aimpg-0.4.2/aimpg/replay/sandbox.py +324 -0
  10. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/verify.py +103 -7
  11. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/workspace.py +4 -1
  12. {aimpg-0.4.0 → aimpg-0.4.2}/docs/STRATEGY.md +1 -1
  13. aimpg-0.4.2/docs/designs/scoreboard-brief.md +46 -0
  14. {aimpg-0.4.0 → aimpg-0.4.2}/pyproject.toml +1 -1
  15. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_e2e.py +3 -3
  16. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_sandbox.py +33 -14
  17. aimpg-0.4.2/tests/test_sandbox_required.py +9 -0
  18. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_verify.py +52 -0
  19. {aimpg-0.4.0 → aimpg-0.4.2}/uv.lock +1 -1
  20. aimpg-0.4.0/.github/workflows/test.yml +0 -17
  21. aimpg-0.4.0/TODOS.md +0 -13
  22. aimpg-0.4.0/aimpg/replay/sandbox.py +0 -175
  23. {aimpg-0.4.0 → aimpg-0.4.2}/.github/workflows/publish.yml +0 -0
  24. {aimpg-0.4.0 → aimpg-0.4.2}/.gitignore +0 -0
  25. {aimpg-0.4.0 → aimpg-0.4.2}/LICENSE +0 -0
  26. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/__init__.py +0 -0
  27. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/attribution.py +0 -0
  28. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/cli.py +0 -0
  29. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/coach.py +0 -0
  30. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/codex_logs.py +0 -0
  31. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/cost.py +0 -0
  32. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/cursor_usage.py +0 -0
  33. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/durable.py +0 -0
  34. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/energy.py +0 -0
  35. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/equivalence.py +0 -0
  36. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/equivalences.json +0 -0
  37. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/factors.json +0 -0
  38. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/gitkept.py +0 -0
  39. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/ledger.py +0 -0
  40. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/live.py +0 -0
  41. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/logs.py +0 -0
  42. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/model.py +0 -0
  43. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/prices.json +0 -0
  44. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/receipt.py +0 -0
  45. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/__init__.py +0 -0
  46. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/hints.py +0 -0
  47. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/picker.py +0 -0
  48. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/select.py +0 -0
  49. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/setups.py +0 -0
  50. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/stats.py +0 -0
  51. {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/share.py +0 -0
  52. {aimpg-0.4.0 → aimpg-0.4.2}/docs/designs/aimpg-design.md +0 -0
  53. {aimpg-0.4.0 → aimpg-0.4.2}/evals/attribution_eval.py +0 -0
  54. {aimpg-0.4.0 → aimpg-0.4.2}/evals/label.py +0 -0
  55. {aimpg-0.4.0 → aimpg-0.4.2}/spikes/agent/RESULTS.md +0 -0
  56. {aimpg-0.4.0 → aimpg-0.4.2}/spikes/agent/allowlist_proxy.py +0 -0
  57. {aimpg-0.4.0 → aimpg-0.4.2}/spikes/agent/make_profile.py +0 -0
  58. {aimpg-0.4.0 → aimpg-0.4.2}/spikes/egress/RESULTS.md +0 -0
  59. {aimpg-0.4.0 → aimpg-0.4.2}/spikes/egress/allowlist_proxy.py +0 -0
  60. {aimpg-0.4.0 → aimpg-0.4.2}/spikes/egress/replay.sb +0 -0
  61. {aimpg-0.4.0 → aimpg-0.4.2}/tests/conftest.py +0 -0
  62. {aimpg-0.4.0 → aimpg-0.4.2}/tests/gitrepo.py +0 -0
  63. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_attribution.py +0 -0
  64. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_coach.py +0 -0
  65. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_codex_logs.py +0 -0
  66. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_cost_and_equivalence.py +0 -0
  67. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_cursor_usage.py +0 -0
  68. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_durable.py +0 -0
  69. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_energy.py +0 -0
  70. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_eval_scoring.py +0 -0
  71. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_gitkept.py +0 -0
  72. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_live.py +0 -0
  73. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_logs.py +0 -0
  74. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_perf.py +0 -0
  75. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_cli.py +0 -0
  76. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_hints.py +0 -0
  77. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_picker.py +0 -0
  78. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_proxy.py +0 -0
  79. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_stats.py +0 -0
  80. {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_share.py +0 -0
@@ -0,0 +1,34 @@
1
+ name: test
2
+ on:
3
+ push:
4
+ branches: [main, "linux-*"]
5
+ pull_request:
6
+ jobs:
7
+ test:
8
+ strategy:
9
+ matrix:
10
+ os: [ubuntu-latest, macos-latest]
11
+ python: ["3.10", "3.13"]
12
+ runs-on: ${{ matrix.os }}
13
+ steps:
14
+ - uses: actions/checkout@v4
15
+ - uses: astral-sh/setup-uv@v6
16
+ - run: git config --global user.email ci@example.com && git config --global user.name CI
17
+ - run: uv run --python ${{ matrix.python }} --group dev pytest -q
18
+ linux-sandbox:
19
+ # Replays on Linux: bubblewrap + the AppArmor rule Ubuntu 24.04 needs for bwrap only
20
+ runs-on: ubuntu-24.04
21
+ steps:
22
+ - uses: actions/checkout@v4
23
+ - uses: astral-sh/setup-uv@v6
24
+ - run: sudo apt-get update -q && sudo apt-get install -y -q bubblewrap
25
+ - name: allow user namespaces for bwrap only
26
+ run: |
27
+ printf 'abi <abi/4.0>,\ninclude <tunables/global>\n\nprofile bwrap /usr/bin/bwrap flags=(unconfined) {\n userns,\n include if exists <local/bwrap>\n}\n' | sudo tee /etc/apparmor.d/bwrap
28
+ sudo apparmor_parser -r /etc/apparmor.d/bwrap
29
+ - run: git config --global user.email ci@example.com && git config --global user.name CI
30
+ - name: an SSH agent for the sandbox to fail to reach
31
+ run: ssh-agent -a /tmp/aimpg-ci-agent.sock && echo "SSH_AUTH_SOCK=/tmp/aimpg-ci-agent.sock" >> "$GITHUB_ENV"
32
+ - run: uv run --group dev pytest -q -rs tests/test_sandbox_required.py tests/test_replay_sandbox.py tests/test_replay_e2e.py
33
+ env:
34
+ AIMPG_REQUIRE_SANDBOX: "1"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: aimpg
3
- Version: 0.4.0
3
+ Version: 0.4.2
4
4
  Summary: Real-world energy per solved task for AI coding agents
5
5
  Project-URL: Homepage, https://github.com/kumarganduri/aimpg
6
6
  Project-URL: Issues, https://github.com/kumarganduri/aimpg/issues
@@ -113,7 +113,7 @@ VERDICT: NOT PROVEN
113
113
  ```
114
114
  That's the real result on the author's Legwork repo: no saving shown, far from the advertised 60–90%. Your repo may differ, and now you can check.
115
115
 
116
- Every run writes a **record** (`~/.aimpg/verify/*.record.json`) holding versions, both setups, each run's tokens, $ and outcome, and the verdict, but never code, diffs or commit messages. Repo and commits are hashed unless you add `--public`. Anyone can recheck the math for free with `aimpg verify --check record.json`, and rerun a public record on their own machine with `aimpg verify --rerun record.json --repo <clone>`.
116
+ Every run writes a **record** (`~/.aimpg/verify/*.record.json`) holding versions, both setups, each run's tokens, $ and outcome, and the verdict, but never code, diffs or commit messages. Repo and commits are hashed unless you add `--public`. Anyone can recheck the math and the record's consistency for free with `aimpg verify --check record.json` (it can't prove the runs happened), and rerun a public record on their own machine with `aimpg verify --rerun record.json --repo <clone>`, which first shows every command, hook and prompt the record would run and asks before running them. Private records keep only fingerprints (SHA-256) of your prompts, CLAUDE.md, settings and commands, never the text. Records also list every redone attempt and excluded commit.
117
117
 
118
118
  Same model: the energy test decides (it must hold at every corner of the energy ranges). Different models or agents: cost per solved task decides, because model sizes are secret. A challenger that solves more than 10 points fewer tasks is never "supported". Codex runs are measured from its own logs and priced from OpenAI's published prices. Codex logs in with an OpenAI **API key** (a ChatGPT subscription login can't be used inside the sandbox); the key is written to the run's throwaway folder and deleted when the run ends. Other commands are judged on solve rate and time only.
119
119
 
@@ -143,11 +143,11 @@ One row per AI-assisted commit: date, repo, energy range, CO₂ range and cost.
143
143
 
144
144
  **Energy:** a physical formula, not a price proxy. Prefill compute for new input, one KV-cache re-read per output token (so long contexts cost more), plus datacenter overhead. Model sizes aren't public, so every number is a range. Overheads follow Google's full-stack measurement of a median Gemini prompt (chips, host CPU and memory, idle capacity, cooling ≈ 1.7× the chips alone), and a chat-sized prompt on a mid-size model comes out at 0.017–0.34 Wh, bracketing Google's disclosed 0.24 Wh ([arXiv 2508.15734](https://arxiv.org/pdf/2508.15734)). Sources for every factor are in [`aimpg/factors.json`](aimpg/factors.json), [`aimpg/prices.json`](aimpg/prices.json) and [`aimpg/equivalences.json`](aimpg/equivalences.json).
145
145
 
146
- **Replays:** each run starts from the code just before your commit, in a fresh, history-free copy, inside a macOS sandbox. The only network allowed is the model API, through an allowlisting proxy. Your own tests judge the result, and edited tests are always restored before judging. Every token count is cross-checked against Claude Code's own totals. Details are in [docs/designs/aimpg-design.md](docs/designs/aimpg-design.md).
146
+ **Replays:** each run starts from the code just before your commit, in a fresh, history-free copy, inside a sandbox (Seatbelt on macOS, bubblewrap on Linux). The only network allowed is the model API, through an allowlisting proxy. Your own tests judge the result, and edited tests are always restored before judging. Every token count is cross-checked against Claude Code's own totals. Details are in [docs/designs/aimpg-design.md](docs/designs/aimpg-design.md).
147
147
 
148
148
  ## Limits (honest list)
149
149
 
150
- - Claude Code and Codex CLI logs are read automatically. Cursor keeps token usage on its servers, so it needs its usage export (below), and that export doesn't say which folder the work was in: give `--cursor-repo` and requests are matched to your next own commit there by time alone (shown as "by time only"). Codex's model prices aren't in the table yet; its requests count toward energy but are listed as unpriced. Replays need macOS and an Anthropic API key (about $0.10–0.30 per run).
150
+ - Claude Code and Codex CLI logs are read automatically. Cursor keeps token usage on its servers, so it needs its usage export (below), and that export doesn't say which folder the work was in: give `--cursor-repo` and requests are matched to your next own commit there by time alone (shown as "by time only"). Replays and `verify` need macOS or Linux and an API key (about $0.10–0.30 per run). On Linux install bubblewrap (`sudo apt install bubblewrap`); Ubuntu 24.04+ also needs a one-time AppArmor rule for bwrap, which aimpg prints if it's missing.
151
151
  - Energy is an estimate with a wide range; dollars are close (within about 7% of Claude Code's own session totals on the author's logs).
152
152
  - Tips are upper bounds and overlap; they can't be added together.
153
153
  - Replays from commit messages alone are hard (12% solved on the author's repo); `--task-mode tests` shows the agent the tests, which makes tasks easier than real work but keeps comparisons fair.
@@ -93,7 +93,7 @@ VERDICT: NOT PROVEN
93
93
  ```
94
94
  That's the real result on the author's Legwork repo: no saving shown, far from the advertised 60–90%. Your repo may differ, and now you can check.
95
95
 
96
- Every run writes a **record** (`~/.aimpg/verify/*.record.json`) holding versions, both setups, each run's tokens, $ and outcome, and the verdict, but never code, diffs or commit messages. Repo and commits are hashed unless you add `--public`. Anyone can recheck the math for free with `aimpg verify --check record.json`, and rerun a public record on their own machine with `aimpg verify --rerun record.json --repo <clone>`.
96
+ Every run writes a **record** (`~/.aimpg/verify/*.record.json`) holding versions, both setups, each run's tokens, $ and outcome, and the verdict, but never code, diffs or commit messages. Repo and commits are hashed unless you add `--public`. Anyone can recheck the math and the record's consistency for free with `aimpg verify --check record.json` (it can't prove the runs happened), and rerun a public record on their own machine with `aimpg verify --rerun record.json --repo <clone>`, which first shows every command, hook and prompt the record would run and asks before running them. Private records keep only fingerprints (SHA-256) of your prompts, CLAUDE.md, settings and commands, never the text. Records also list every redone attempt and excluded commit.
97
97
 
98
98
  Same model: the energy test decides (it must hold at every corner of the energy ranges). Different models or agents: cost per solved task decides, because model sizes are secret. A challenger that solves more than 10 points fewer tasks is never "supported". Codex runs are measured from its own logs and priced from OpenAI's published prices. Codex logs in with an OpenAI **API key** (a ChatGPT subscription login can't be used inside the sandbox); the key is written to the run's throwaway folder and deleted when the run ends. Other commands are judged on solve rate and time only.
99
99
 
@@ -123,11 +123,11 @@ One row per AI-assisted commit: date, repo, energy range, CO₂ range and cost.
123
123
 
124
124
  **Energy:** a physical formula, not a price proxy. Prefill compute for new input, one KV-cache re-read per output token (so long contexts cost more), plus datacenter overhead. Model sizes aren't public, so every number is a range. Overheads follow Google's full-stack measurement of a median Gemini prompt (chips, host CPU and memory, idle capacity, cooling ≈ 1.7× the chips alone), and a chat-sized prompt on a mid-size model comes out at 0.017–0.34 Wh, bracketing Google's disclosed 0.24 Wh ([arXiv 2508.15734](https://arxiv.org/pdf/2508.15734)). Sources for every factor are in [`aimpg/factors.json`](aimpg/factors.json), [`aimpg/prices.json`](aimpg/prices.json) and [`aimpg/equivalences.json`](aimpg/equivalences.json).
125
125
 
126
- **Replays:** each run starts from the code just before your commit, in a fresh, history-free copy, inside a macOS sandbox. The only network allowed is the model API, through an allowlisting proxy. Your own tests judge the result, and edited tests are always restored before judging. Every token count is cross-checked against Claude Code's own totals. Details are in [docs/designs/aimpg-design.md](docs/designs/aimpg-design.md).
126
+ **Replays:** each run starts from the code just before your commit, in a fresh, history-free copy, inside a sandbox (Seatbelt on macOS, bubblewrap on Linux). The only network allowed is the model API, through an allowlisting proxy. Your own tests judge the result, and edited tests are always restored before judging. Every token count is cross-checked against Claude Code's own totals. Details are in [docs/designs/aimpg-design.md](docs/designs/aimpg-design.md).
127
127
 
128
128
  ## Limits (honest list)
129
129
 
130
- - Claude Code and Codex CLI logs are read automatically. Cursor keeps token usage on its servers, so it needs its usage export (below), and that export doesn't say which folder the work was in: give `--cursor-repo` and requests are matched to your next own commit there by time alone (shown as "by time only"). Codex's model prices aren't in the table yet; its requests count toward energy but are listed as unpriced. Replays need macOS and an Anthropic API key (about $0.10–0.30 per run).
130
+ - Claude Code and Codex CLI logs are read automatically. Cursor keeps token usage on its servers, so it needs its usage export (below), and that export doesn't say which folder the work was in: give `--cursor-repo` and requests are matched to your next own commit there by time alone (shown as "by time only"). Replays and `verify` need macOS or Linux and an API key (about $0.10–0.30 per run). On Linux install bubblewrap (`sudo apt install bubblewrap`); Ubuntu 24.04+ also needs a one-time AppArmor rule for bwrap, which aimpg prints if it's missing.
131
131
  - Energy is an estimate with a wide range; dollars are close (within about 7% of Claude Code's own session totals on the author's logs).
132
132
  - Tips are upper bounds and overlap; they can't be added together.
133
133
  - Replays from commit messages alone are hard (12% solved on the author's repo); `--task-mode tests` shows the agent the tests, which makes tasks easier than real work but keeps comparisons fair.
aimpg-0.4.2/TODOS.md ADDED
@@ -0,0 +1,9 @@
1
+ # TODOS
2
+
3
+ ## Phase 2
4
+
5
+ The earlier TODOs (test leakage + memorization, statistics gate, network-limit spike) were resolved by the Phase 2 eng review (R3/R4, R7/R15/R16, spikes/egress + spikes/agent).
6
+
7
+ ### ~~Linux replay sandbox~~ (done 2026-10-05)
8
+ bubblewrap backend in aimpg/replay/sandbox.py; the proxy is relayed into the sandbox through a Unix socket. CI's `linux-sandbox` job runs the escape and end-to-end tests on Ubuntu 24.04.
9
+ - **Still untested on Linux:** a real paid Claude Code or Codex run (CI has no API keys); only the stand-in agent has run there.
@@ -21,10 +21,11 @@ from aimpg.replay import run as runner
21
21
  from aimpg.replay import select, stats
22
22
  from aimpg.replay.proxy import AllowlistProxy
23
23
  from aimpg.replay import picker
24
+ from aimpg.replay import sandbox
24
25
  from aimpg.replay.setups import SETUPS, with_model
25
26
  from aimpg.replay.workspace import Layout
26
27
 
27
- ROOT = Path("/private/tmp/aimpg-replay") # outside $HOME; every agent profile denies it
28
+ ROOT = sandbox.TMP / "aimpg-replay" # outside $HOME; every agent profile denies it
28
29
  STATE = Path.home() / ".aimpg" / "replay" # selections + results, unreadable to agents
29
30
  DEFAULT_MODEL = "claude-sonnet-5-5"
30
31
  DEFAULT_SETUPS = "claude-code,claude-code+terse,claude-code+rtk"
@@ -53,18 +53,23 @@ def transcript(cfg: Path, cwd: Path, model: str, n: int) -> tuple[dict, dict]:
53
53
 
54
54
 
55
55
  def escape_attempts(cwd: Path, cfg: Path) -> list[str]:
56
- """Every attempt that SUCCEEDS is a sandbox bug."""
56
+ """Every attempt that SUCCEEDS is a sandbox bug.
57
+
58
+ Seeing a folder counts only if real content shows: on Linux, home and the
59
+ runs folder are empty private stand-ins, which is the sandbox working.
60
+ """
57
61
  succeeded = []
58
62
  real_home = Path(pwd.getpwuid(os.getuid()).pw_dir)
59
- for target in (real_home, real_home / ".ssh", cwd.parent.parent): # home, keys, sibling runs
63
+ own_run = cwd.parent.name
64
+ for target, allowed in ((real_home, set()), (real_home / ".ssh", set()), (cwd.parent.parent, {own_run})):
60
65
  try:
61
- os.listdir(target)
62
- succeeded.append(f"read {target}")
66
+ if set(os.listdir(target)) - allowed:
67
+ succeeded.append(f"read {target}")
63
68
  except OSError:
64
69
  pass
65
70
  try:
66
- Path("/private/tmp/aimpg-escape-probe").write_text("x")
67
- succeeded.append("write /private/tmp")
71
+ (real_home / "aimpg-escape-probe").write_text("x")
72
+ succeeded.append("write to home")
68
73
  except OSError:
69
74
  pass
70
75
  probe = subprocess.run(["/usr/bin/curl", "-s", "-m", "5", "-o", "/dev/null", "-w", "%{http_code}", "--noproxy", "*", "https://github.com"], capture_output=True, text=True)
@@ -15,7 +15,11 @@ The allowlist changes per phase:
15
15
  from __future__ import annotations
16
16
 
17
17
  import asyncio
18
+ import shutil
19
+ import sys
20
+ import tempfile
18
21
  import threading
22
+ from pathlib import Path
19
23
  from dataclasses import dataclass, field
20
24
 
21
25
  PYPI = frozenset({"pypi.org", "files.pythonhosted.org"})
@@ -33,6 +37,9 @@ class AllowlistProxy:
33
37
  port: int = 0
34
38
  upstream_port: int = 443 # tests point this at a local echo server
35
39
  log: list[tuple[str, str]] = field(default_factory=list) # ("ALLOW" | "BLOCK", "host:port")
40
+ # Linux: also a Unix socket, bound into the sandbox (its private network can't reach host ports)
41
+ socket_path: Path | None = None
42
+ unix: bool = field(default_factory=lambda: sys.platform.startswith("linux"))
36
43
  _loop: asyncio.AbstractEventLoop | None = None
37
44
  _thread: threading.Thread | None = None
38
45
  _ready: threading.Event = field(default_factory=threading.Event)
@@ -49,6 +56,8 @@ class AllowlistProxy:
49
56
  self._loop.call_soon_threadsafe(self._loop.stop)
50
57
  if self._thread is not None:
51
58
  self._thread.join(5)
59
+ if self.socket_path is not None:
60
+ shutil.rmtree(self.socket_path.parent, ignore_errors=True)
52
61
 
53
62
  def set_allow(self, hosts: frozenset[str]) -> None:
54
63
  self.allow = frozenset(hosts)
@@ -65,16 +74,23 @@ class AllowlistProxy:
65
74
  asyncio.set_event_loop(self._loop)
66
75
  server = self._loop.run_until_complete(asyncio.start_server(self._handle, "127.0.0.1", self.port))
67
76
  self.port = server.sockets[0].getsockname()[1]
77
+ servers = [server]
78
+ if self.unix:
79
+ # its own folder: the sandbox gets only this folder, nothing else from /tmp
80
+ self.socket_path = Path(tempfile.mkdtemp(prefix="aimpg-proxy-", dir="/tmp")) / "proxy.sock"
81
+ servers.append(self._loop.run_until_complete(asyncio.start_unix_server(self._handle, str(self.socket_path))))
68
82
  self._ready.set()
69
83
  try:
70
84
  self._loop.run_forever()
71
85
  finally:
72
- server.close()
86
+ for srv in servers:
87
+ srv.close()
73
88
  pending = [t for t in asyncio.all_tasks(self._loop) if not t.done()]
74
89
  for task in pending:
75
90
  task.cancel()
76
91
  self._loop.run_until_complete(asyncio.gather(*pending, return_exceptions=True))
77
- self._loop.run_until_complete(server.wait_closed())
92
+ for srv in servers:
93
+ self._loop.run_until_complete(srv.wait_closed())
78
94
  self._loop.close()
79
95
 
80
96
  async def _handle(self, reader: asyncio.StreamReader, writer: asyncio.StreamWriter) -> None:
@@ -215,6 +215,7 @@ def _run_agent(commit: Commit, setup: Setup, layout: Layout, cfg: Config, work:
215
215
  # the source repo too: it holds the answer, and may live outside $HOME
216
216
  deny_roots=[layout.root, Path(commit.repo)],
217
217
  proxy_port=proxy.port,
218
+ proxy_socket=proxy.socket_path,
218
219
  )
219
220
  res = sandbox.run(
220
221
  setup.argv(task_text(commit, cfg.task_mode), cfg.model, cfg.per_run_budget_usd, cfgdir),
@@ -443,6 +444,17 @@ def load_results(path: Path) -> tuple[list[Record], set[str]]:
443
444
  return final, excluded
444
445
 
445
446
 
447
+ def all_attempts(path: Path) -> list[Record]:
448
+ """Every run line in a results file, including retries and redone runs (for records)."""
449
+ out = []
450
+ for line in path.read_text().splitlines():
451
+ if line.strip():
452
+ data = json.loads(line)
453
+ if "excluded_commit" not in data:
454
+ out.append(Record(**data))
455
+ return out
456
+
457
+
446
458
  def make_solution(commit: Commit, dest: Path) -> Path:
447
459
  """Tar of the commit's non-test files (fake 'solve' agent only)."""
448
460
  tmp = dest.with_suffix(".tree")
@@ -0,0 +1,324 @@
1
+ """macOS Seatbelt sandbox for replay runs.
2
+
3
+ Read rules adapted from Legwork's hardened profile
4
+ (github.com/kumarganduri/legwork, legwork/sandbox_runner.py `_generate_profile`,
5
+ MIT, same author): reads are allowed outside $HOME (system libraries) but not
6
+ under it; the workdir is re-allowed. Added for replays:
7
+
8
+ * network only to the local allowlisting proxy port (or none at all);
9
+ * a deny-all replay root, re-allowing only this run's own folders, so parallel
10
+ runs can't read each other's hidden tests (verified: later, more specific
11
+ rules win);
12
+ * read-only access to the agent's and toolchain's own install folders under
13
+ $HOME (Claude Code, uv, node), found by resolving the binaries.
14
+
15
+ Linux: the same policy through bubblewrap (adapted from Legwork's
16
+ `_bwrap_args`): the system read-only; home folders, /tmp, /var/tmp and /run
17
+ replaced by empty private tmpfs (so other runs, the answer repo and local
18
+ sockets such as the SSH agent are gone); only this run's folders bound back
19
+ writable, the agent's install folders read-only. Every namespace is
20
+ unshared, network included. The allowlisting proxy can't be reached as a
21
+ host port from a private network namespace, so it also listens on a Unix
22
+ socket that is bound in, and a tiny relay inside the sandbox listens on
23
+ 127.0.0.1:<the same port> and forwards to it. Ubuntu 24.04+ needs an
24
+ AppArmor profile that lets bwrap (only) create user namespaces; the error
25
+ message prints it.
26
+
27
+ Other platforms raise SandboxUnavailable.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import functools
33
+ import json
34
+ import os
35
+ import shutil
36
+ import signal
37
+ import subprocess
38
+ import sys
39
+ from dataclasses import dataclass, field
40
+ from pathlib import Path
41
+
42
+ _MACH_SERVICES = (
43
+ "com.apple.system.opendirectoryd.libinfo",
44
+ "com.apple.system.opendirectoryd.membership",
45
+ "com.apple.system.notification_center",
46
+ "com.apple.system.logger",
47
+ "com.apple.logd",
48
+ "com.apple.diagnosticd",
49
+ "com.apple.SystemConfiguration.configd",
50
+ "com.apple.SystemConfiguration.DNSConfiguration",
51
+ "com.apple.dnssd.service",
52
+ "com.apple.trustd",
53
+ "com.apple.trustd.agent",
54
+ "com.apple.cfprefsd.daemon",
55
+ "com.apple.cfprefsd.agent",
56
+ )
57
+ # Where replay data lives: outside $HOME on both systems.
58
+ TMP = Path("/private/tmp") if sys.platform == "darwin" else Path("/tmp")
59
+ TOOLS = ("claude", "codex", "uv", "node", "npm", "npx", "git", "rtk")
60
+
61
+
62
+ class SandboxUnavailable(RuntimeError):
63
+ pass
64
+
65
+
66
+ def available() -> bool:
67
+ return unavailable_reason() is None
68
+
69
+
70
+ BWRAP_APPARMOR_PROFILE = """abi <abi/4.0>,
71
+ include <tunables/global>
72
+
73
+ profile bwrap /usr/bin/bwrap flags=(unconfined) {
74
+ userns,
75
+ include if exists <local/bwrap>
76
+ }
77
+ """
78
+
79
+
80
+ def unavailable_reason() -> str | None:
81
+ """Why replays can't be sandboxed here, with the fix; None if they can."""
82
+ if sys.platform == "darwin":
83
+ return None if _launcher("sandbox-exec") else "sandbox-exec is missing (it ships with macOS)"
84
+ if sys.platform.startswith("linux"):
85
+ if not _launcher("bwrap"):
86
+ return "replays on Linux need bubblewrap: sudo apt install bubblewrap (or dnf/pacman)"
87
+ error = _bwrap_probe()
88
+ if error is None:
89
+ return None
90
+ hint = ""
91
+ try:
92
+ if Path("/proc/sys/kernel/apparmor_restrict_unprivileged_userns").read_text().strip() == "1":
93
+ hint = ("\nUbuntu 24.04+ restricts user namespaces with AppArmor; allow them for bwrap only:\n"
94
+ " sudo tee /etc/apparmor.d/bwrap <<'EOF'\n" + BWRAP_APPARMOR_PROFILE + "EOF\n"
95
+ " sudo apparmor_parser -r /etc/apparmor.d/bwrap")
96
+ except OSError:
97
+ pass
98
+ return f"bubblewrap can't create a sandbox here ({error}){hint}"
99
+ return f"replays need macOS or Linux (not {sys.platform})"
100
+
101
+
102
+ @functools.lru_cache(maxsize=None)
103
+ def _launcher(name: str) -> str | None:
104
+ """Absolute path of the sandbox launcher, looked up in system folders only,
105
+ never in the child's PATH (which a sandboxed step could plant a fake in)."""
106
+ found = shutil.which(name, path="/usr/bin:/bin:/usr/sbin:/sbin:/usr/local/bin")
107
+ return os.path.realpath(found) if found else None
108
+
109
+
110
+ @functools.lru_cache(maxsize=1)
111
+ def _bwrap_probe() -> str | None:
112
+ try:
113
+ r = subprocess.run([_launcher("bwrap"), "--unshare-all", "--ro-bind", "/", "/", "--dev", "/dev", "--proc", "/proc", "true"],
114
+ stdin=subprocess.DEVNULL, capture_output=True, text=True, timeout=15)
115
+ except (OSError, subprocess.TimeoutExpired) as exc:
116
+ return str(exc)
117
+ return None if r.returncode == 0 else (r.stderr.strip() or f"exit {r.returncode}")
118
+
119
+
120
+ def tool_dirs(tools: tuple[str, ...] = TOOLS) -> list[Path]:
121
+ """Install folders of the agent/toolchain that live under $HOME (read-only grants).
122
+
123
+ For a binary in some `bin/`, its package root (one level up) is granted so
124
+ node/npm can read their libraries, but never $HOME or ~/.local themselves:
125
+ uv sits directly in ~/.local/bin, and ~/.local holds other apps' data.
126
+ """
127
+ home = Path.home().resolve()
128
+ too_broad = {home, home / ".local", home / ".local" / "share"}
129
+ found: set[Path] = set()
130
+ for name in tools:
131
+ which = shutil.which(name)
132
+ if not which:
133
+ continue
134
+ for p in (Path(which).absolute(), Path(which).resolve()):
135
+ d = p.parent
136
+ if d.name in ("bin", "versions") and d.parent not in too_broad:
137
+ d = d.parent
138
+ if str(d).startswith(str(home) + os.sep) and d not in too_broad:
139
+ found.add(d)
140
+ uv_python = home / ".local/share/uv/python"
141
+ if uv_python.is_dir():
142
+ found.add(uv_python.resolve())
143
+ return sorted(found)
144
+
145
+
146
+ def tool_path(tools: tuple[str, ...] = TOOLS) -> str:
147
+ """PATH made of the folders the tools are actually invoked from."""
148
+ dirs: list[str] = []
149
+ for name in tools:
150
+ which = shutil.which(name)
151
+ if which and str(Path(which).parent) not in dirs and str(Path(which).parent) not in ("/usr/bin", "/bin"):
152
+ dirs.append(str(Path(which).parent))
153
+ return ":".join(dirs + ["/usr/bin", "/bin", "/usr/sbin", "/sbin"])
154
+
155
+
156
+ @dataclass
157
+ class Profile:
158
+ writable: list[Path]
159
+ readable: list[Path] = field(default_factory=list)
160
+ deny_roots: list[Path] = field(default_factory=list)
161
+ proxy_port: int | None = None # None: no network at all
162
+ proxy_socket: Path | None = None # Linux: the proxy's Unix socket, relayed to 127.0.0.1:proxy_port inside
163
+
164
+ def render(self) -> str:
165
+ home = str(Path.home().resolve())
166
+ writable = [str(Path(p).resolve()) for p in self.writable]
167
+ readable = [str(Path(p).resolve()) for p in self.readable]
168
+ lines = [
169
+ "(version 1)",
170
+ "(deny default)",
171
+ "(allow process-fork)",
172
+ "(allow process-exec)",
173
+ "(allow sysctl-read)",
174
+ "(allow mach-lookup " + " ".join(f'(global-name "{m}")' for m in _MACH_SERVICES) + ")",
175
+ "(allow iokit-open)",
176
+ # Codex syncs macOS managed preferences at startup and refuses to run
177
+ # without them; cfprefsd shares them read-only through this memory.
178
+ '(allow ipc-posix-shm-read* (ipc-posix-name-prefix "apple.cfprefs"))',
179
+ "(allow signal (target same-sandbox))",
180
+ f'(allow file-read* (require-not (subpath "{home}")))',
181
+ ]
182
+ for root in self.deny_roots:
183
+ lines.append(f'(deny file-read* file-write* (subpath "{Path(root).resolve()}"))')
184
+ for d in readable:
185
+ lines.append(f'(allow file-read* (subpath "{d}"))')
186
+ for d in writable:
187
+ lines.append(f'(allow file-read* file-write* (subpath "{d}"))')
188
+ for ancestor in sorted(_ancestors([*writable, *readable], [home, *map(str, self.deny_roots)])):
189
+ lines.append(f'(allow file-read-metadata (literal "{ancestor}"))')
190
+ lines.append('(allow file-write-data (literal "/dev/null") (literal "/dev/zero"))')
191
+ if self.proxy_port is not None:
192
+ lines += ["(allow system-socket)", f'(allow network-outbound (remote ip "localhost:{self.proxy_port}"))']
193
+ return "\n".join(lines) + "\n"
194
+
195
+
196
+ def bwrap_args(profile: Profile, cwd: Path) -> list[str]:
197
+ """Linux equivalent of Profile.render (see the module docstring)."""
198
+ home = Path.home().resolve()
199
+ hidden = sorted({p for p in (Path("/home"), Path("/root"), home) if p.is_dir()}, key=lambda p: len(p.parts))
200
+ args = [_launcher("bwrap") or "/usr/bin/bwrap", "--unshare-all", "--die-with-parent", "--new-session",
201
+ "--ro-bind", "/", "/", "--dev", "/dev", "--proc", "/proc",
202
+ "--tmpfs", "/tmp", "--tmpfs", "/var/tmp", "--tmpfs", "/run"]
203
+ if os.path.isdir("/var/run") and not os.path.islink("/var/run"):
204
+ args += ["--tmpfs", "/var/run"]
205
+ for d in hidden:
206
+ args += ["--tmpfs", str(d)]
207
+ for root in profile.deny_roots: # e.g. the answer repo, wherever it lives
208
+ r = Path(root).resolve()
209
+ if r.is_dir() and not any(r == h or h in r.parents for h in (*hidden, Path("/tmp"), Path("/var/tmp"))):
210
+ args += ["--tmpfs", str(r)]
211
+ for d in profile.readable:
212
+ d = Path(d).resolve()
213
+ if d.exists():
214
+ args += ["--ro-bind", str(d), str(d)]
215
+ if profile.proxy_socket is not None:
216
+ python = Path(sys.base_prefix).resolve() # the relay runs on this Python's stdlib
217
+ args += ["--ro-bind", str(python), str(python)]
218
+ sock_dir = Path(profile.proxy_socket).parent.resolve()
219
+ args += ["--bind", str(sock_dir), str(sock_dir)]
220
+ for d in profile.writable:
221
+ d = Path(d).resolve()
222
+ args += ["--bind", str(d), str(d)]
223
+ for d in reversed(hidden): # stand-in homes read-only: writes fail as on macOS
224
+ args += ["--remount-ro", str(d)]
225
+ args += ["--chdir", str(Path(cwd).resolve())]
226
+ return args
227
+
228
+
229
+ # Runs inside the Linux sandbox as `python -c`: 127.0.0.1:PORT -> the proxy's Unix socket.
230
+ _RELAY = """
231
+ import socket, subprocess, sys, threading
232
+ sock_path, port, argv = sys.argv[1], int(sys.argv[2]), sys.argv[4:]
233
+ def pipe(a, b):
234
+ try:
235
+ while True:
236
+ data = a.recv(65536)
237
+ if not data:
238
+ break
239
+ b.sendall(data)
240
+ except OSError:
241
+ pass
242
+ finally:
243
+ for s in (a, b):
244
+ try:
245
+ s.shutdown(socket.SHUT_RDWR)
246
+ except OSError:
247
+ pass
248
+ def serve(srv):
249
+ while True:
250
+ client, _ = srv.accept()
251
+ try:
252
+ up = socket.socket(socket.AF_UNIX)
253
+ up.connect(sock_path)
254
+ except OSError:
255
+ client.close()
256
+ continue
257
+ threading.Thread(target=pipe, args=(client, up), daemon=True).start()
258
+ threading.Thread(target=pipe, args=(up, client), daemon=True).start()
259
+ srv = socket.socket()
260
+ srv.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
261
+ srv.bind(("127.0.0.1", port))
262
+ srv.listen(64)
263
+ threading.Thread(target=serve, args=(srv,), daemon=True).start()
264
+ code = subprocess.call(argv)
265
+ sys.exit(code if code >= 0 else 128 - code)
266
+ """
267
+
268
+
269
+ def _ancestors(paths: list[str], roots: list[str]) -> set[str]:
270
+ """Parent dirs between a denied root and an allowed path: metadata only, so realpath() works."""
271
+ out = set()
272
+ for p in paths:
273
+ for root in roots:
274
+ root = str(Path(root).resolve())
275
+ if p.startswith(root + os.sep):
276
+ cur = Path(p).parent
277
+ while str(cur).startswith(root):
278
+ out.add(str(cur))
279
+ if str(cur) == root:
280
+ break
281
+ cur = cur.parent
282
+ return out
283
+
284
+
285
+ @dataclass
286
+ class Result:
287
+ returncode: int | None # None when timed out
288
+ stdout: str
289
+ stderr: str
290
+ timed_out: bool = False
291
+
292
+
293
+ def run(argv: list[str], *, profile: Profile, profile_path: Path, env: dict[str, str], cwd: Path, timeout: float) -> Result:
294
+ """Run argv inside the sandbox. The whole process group is killed on timeout."""
295
+ reason = unavailable_reason()
296
+ if reason:
297
+ raise SandboxUnavailable(reason)
298
+ if sys.platform == "darwin":
299
+ profile_path.write_text(profile.render())
300
+ full = [_launcher("sandbox-exec"), "-f", str(profile_path), *argv]
301
+ else:
302
+ inner = argv
303
+ if profile.proxy_socket is not None:
304
+ inner = [str(Path(sys.executable).resolve()), "-I", "-c", _RELAY, str(profile.proxy_socket), str(profile.proxy_port), "--", *argv]
305
+ full = [*bwrap_args(profile, cwd), "--", *inner]
306
+ profile_path.write_text(json.dumps(full[:-len(argv)] if argv else full, indent=1)) # audit copy
307
+ proc = subprocess.Popen(
308
+ full,
309
+ cwd=cwd,
310
+ env=env,
311
+ stdin=subprocess.DEVNULL, # Codex waits for "additional input" on an open stdin
312
+ stdout=subprocess.PIPE,
313
+ stderr=subprocess.PIPE,
314
+ text=True,
315
+ errors="replace",
316
+ start_new_session=True,
317
+ )
318
+ try:
319
+ out, err = proc.communicate(timeout=timeout)
320
+ return Result(proc.returncode, out, err)
321
+ except subprocess.TimeoutExpired:
322
+ os.killpg(proc.pid, signal.SIGKILL)
323
+ out, err = proc.communicate()
324
+ return Result(None, out, err, timed_out=True)